From 6454aa9904d2e97f2102ae57893f98eccc1ff4ec Mon Sep 17 00:00:00 2001 From: serprex <159546+serprex@users.noreply.github.com> Date: Thu, 1 Oct 2026 15:04:50 +0000 Subject: [PATCH] Enforce ordering invariants with typestate - Require persistence before publishing durable positions or swapping tables - Hold pending rows until swap completes, recover them after restart - Require replay and barrier checks before catalog reads and DDL - Require completed inserts before ack, verified checksums before plan replay - Bind sequence numbers to their ack collector - Separate setup from running state, replace flag combinations with enums - Distinguish unaligned resume positions from retention floors - Fix timeline 1 incorrectly falling back to COPY - Expand coverage for corrupt plans, swap recovery and archive replay --- .github/workflows/ci.yml | 2 +- Cargo.lock | 1 + Cargo.toml | 2 + benches/bootstrap_pump.rs | 4 +- benches/xact_spill.rs | 10 +- plans/coverage100.md | 6 +- src/backfill/backfill_staging.rs | 4 +- src/backfill/backup_backfill.rs | 162 ++++++----- src/backfill/backup_checkpoint.rs | 47 +++- src/backfill/backup_page_walk.rs | 148 ++++++----- src/backfill/bootstrap_marker.rs | 12 +- src/backfill/bootstrap_window.rs | 3 +- src/backfill/copy_backfill.rs | 414 +++++++++++++++++++---------- src/backfill/spool.rs | 256 ++++++++++++------ src/backfill/visibility_gate.rs | 32 ++- src/backfill/visibility_pending.rs | 95 ++++++- src/backfill/wal_landing.rs | 3 +- src/backfill/wal_replay.rs | 57 ++-- src/backfill/walk_barrier.rs | 37 ++- src/bin/stream/bootstrap.rs | 52 ++-- src/bin/stream/housekeeping.rs | 36 +-- src/bin/stream/metrics_publish.rs | 2 +- src/bin/stream/runtime_cfg.rs | 21 +- src/bin/stream/session.rs | 167 ++++++------ src/bin/stream/source_db.rs | 48 ++-- src/bin/stream/source_recovery.rs | 61 +++-- src/catalog/shadow_catalog.rs | 44 ++- src/ch.rs | 47 ++-- src/config.rs | 73 ++--- src/decode/visibility.rs | 32 ++- src/emit/pipeline/ack.rs | 113 ++++++-- src/emit/pipeline/bootstrap.rs | 103 ++++--- src/emit/pipeline/inserter.rs | 55 ++-- src/emit/pipeline/mod.rs | 14 +- src/emit/pipeline/plan_spool.rs | 174 +++++++++++- src/emit/pipeline/planner.rs | 47 +++- src/emit/pipeline/reorder.rs | 191 +++++++------ src/emit/pipeline/tail.rs | 26 +- src/filter/engine.rs | 49 ++-- src/ops/bridge.rs | 83 +++--- src/ops/control.rs | 1 + src/pos.rs | 98 +++++-- src/record.rs | 8 +- src/runtime_config.rs | 2 +- src/source/archive.rs | 15 +- src/source/boundary_hold.rs | 67 +++-- src/source/catalog_capture.rs | 63 ++++- src/source/manifest.rs | 19 +- src/source/queueing_record_sink.rs | 1 - src/source/resume_prefix.rs | 6 +- src/source/segment_sink.rs | 25 +- src/source/shadow_stream.rs | 11 +- src/source/streaming_walker.rs | 37 ++- src/source/transition.rs | 69 +++-- src/source/wal_stream.rs | 104 +++++--- src/source_db.rs | 22 +- src/toast/resolver.rs | 3 +- src/toast/shadow_store.rs | 2 +- src/toast/toast_retire.rs | 1 + src/xact/xact_buffer.rs | 375 +++++++++++++++----------- tests/backfill_staging_e2e.rs | 163 +++++++++++- tests/bootstrap_pipeline_ch.rs | 4 +- tests/bridge.rs | 16 +- tests/common/inproc_harness.rs | 39 +-- tests/emitter_budget_flush.rs | 11 +- tests/emitter_native_types.rs | 11 +- tests/emitter_tls.rs | 11 +- tests/fpi_user_pages.rs | 4 +- tests/multi_segment_filter.rs | 1 - tests/multixact_visibility.rs | 2 +- tests/oracle_types_e2e.rs | 2 +- tests/resume_shadow_relations.rs | 3 +- tests/toast_mode_e2e.rs | 4 +- tests/toast_rewrite_e2e.rs | 4 +- tests/toast_tombstone_e2e.rs | 4 +- tests/toast_truncate_drop_e2e.rs | 4 +- tests/vacuum_catalog_churn.rs | 10 +- tests/wal_stream_chunk_boundary.rs | 3 +- tests/wal_stream_e2e.rs | 3 +- tests/wal_stream_throughput.rs | 4 +- tests/xact_buffer.rs | 83 ++++-- 81 files changed, 2647 insertions(+), 1401 deletions(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 91260cca..065bfbf0 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -283,7 +283,7 @@ jobs: - name: Coverage summary working-directory: walshadow env: - MAX_MISSED_LINES: 3130 + MAX_MISSED_LINES: 2950 run: | awk -F'[:,]' -v max="$MAX_MISSED_LINES" ' /^DA:/ { f++; if ($3 > 0) h++ } diff --git a/Cargo.lock b/Cargo.lock index 21dd88dd..a45ea7ab 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -3091,6 +3091,7 @@ dependencies = [ "tracing-subscriber", "url", "wal-rus", + "walshadow", "zstd", ] diff --git a/Cargo.toml b/Cargo.toml index ca6ffeb6..257a4bb4 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -37,6 +37,7 @@ path = "src/lib.rs" default = ["lz4", "zstd"] lz4 = ["clickhouse-c-rs/lz4"] zstd = ["clickhouse-c-rs/zstd"] +test-support = [] [dependencies] wal-rus = "0.3.5" @@ -101,6 +102,7 @@ name = "spool_read" harness = false [dev-dependencies] +walshadow = { path = ".", features = ["test-support"] } # Pre-main tracing subscriber for tests ctor = "1.0.13" tempfile = "3" diff --git a/benches/bootstrap_pump.rs b/benches/bootstrap_pump.rs index 825b4abd..84af9213 100644 --- a/benches/bootstrap_pump.rs +++ b/benches/bootstrap_pump.rs @@ -401,9 +401,7 @@ async fn bench_drain(shape: Shape) -> Report { .unwrap(); let elapsed = started.elapsed().as_secs_f64(); feeder.await.unwrap(); - drop(msg_tx); - drop(ack); - tail.join().await; + tail.close(msg_tx, ack).await; Report { label: "drain".into(), diff --git a/benches/xact_spill.rs b/benches/xact_spill.rs index 34c41add..29cdbade 100644 --- a/benches/xact_spill.rs +++ b/benches/xact_spill.rs @@ -9,7 +9,7 @@ use clap::Parser; use walrus::pg::walparser::RelFileNode; use walshadow::heap_decoder::{ColumnValue, DecodedHeap, DecodedTuple, DescribedHeap, HeapOp}; use walshadow::schema::{RelDescriptor, RelName, ReplIdent}; -use walshadow::xact_buffer::{XactBuffer, XactBufferConfig}; +use walshadow::xact_buffer::{StashResolved, XactBuffer, XactBufferConfig}; #[global_allocator] static GLOBAL: mimalloc::MiMalloc = mimalloc::MiMalloc; @@ -96,7 +96,13 @@ async fn run(args: Args) -> anyhow::Result<()> { let start = Instant::now(); let commit_lsn = u64::from(xid + 1) * (args.rows as u64 + 1); let mut drain = buffer - .drain_committed(xid, 0, commit_lsn, &[], false) + .drain_committed( + StashResolved::nothing_stashed(xid), + 0, + commit_lsn, + &[], + false, + ) .await?; let mut rows = 0; while let Some(batch) = drain.next_batch(1024, 1 << 20, None).await? { diff --git a/plans/coverage100.md b/plans/coverage100.md index a287f30c..88440798 100644 --- a/plans/coverage100.md +++ b/plans/coverage100.md @@ -136,9 +136,9 @@ Check candidate gaps against fresh merged report before adding tests for `reopen_walk_spools` and ready-checkpoint resume, or hand-seed checkpoints as `backup_checkpoint_e2e` does - `copy_backfill` pending tables: hold an open xact on target table across a - base-backup opt-in, commit or abort after walk, for `record_pending` and - `settle_ended_pending`. `copy_chunk_blocks = 1` on a multi-block table for chunk - loop and progress ledger. Empty table fast path. Opt-out and CH schema change + base-backup opt-in, commit or abort after walk, for `hold_pending`, + `release_pending` and `settle_ended_pending`. `copy_chunk_blocks = 1` on a + multi-block table for chunk loop and progress ledger. Empty table fast path. Opt-out and CH schema change mid-pass for publish/swap edges. Two opt-ins inside coalesce window - `copy_backfill` units: `wire_kind` per OID, `decode_field` per type and invalid UTF-8, ledger version rejection. Ledger persist failures via fs faults diff --git a/src/backfill/backfill_staging.rs b/src/backfill/backfill_staging.rs index 3b28e1fe..799724ad 100644 --- a/src/backfill/backfill_staging.rs +++ b/src/backfill/backfill_staging.rs @@ -22,6 +22,7 @@ use anyhow::{Context, Result, bail}; use clickhouse_c::{Block, Event}; use crate::backfill::backfill_types::BackupRequest; +use crate::backfill::copy_backfill::SwapPermit; use crate::ch::{ChConn, EmitterError, exec_drain, quote_ident, with_timeout}; use crate::config::DestEmitter; use crate::mapping::{MappingHandle, TableMapping, TableTarget}; @@ -271,7 +272,8 @@ impl StagingSession { } /// Atomic publish; requires an Atomic/Replicated database engine. - pub async fn exchange(&mut self, rel: &StagingRel) -> Result<()> { + pub async fn exchange(&mut self, permit: SwapPermit<'_>) -> Result<()> { + let rel = permit.rel(); self.exec_once(&format!( "EXCHANGE TABLES {} AND {}", rel.real_sql(), diff --git a/src/backfill/backup_backfill.rs b/src/backfill/backup_backfill.rs index 75eaab9d..6be9d687 100644 --- a/src/backfill/backup_backfill.rs +++ b/src/backfill/backup_backfill.rs @@ -54,13 +54,14 @@ use walrus::pg::walparser::{Oid, RmId}; use crate::backfill::backfill_types::{BackupRequest, PassContext, PassOutcome}; use crate::backfill::backup_checkpoint::{self, BackupCheckpoint, WalkState}; use crate::backfill::backup_page_walk::{ - BOOTSTRAP_TUPLE_CHANNEL_CAP, BackfillTuple, CatalogMap, PageWalkSink, + BOOTSTRAP_TUPLE_CHANNEL_CAP, BackfillTuple, CatalogMap, CompleteAccums, GateAccums, + PageWalkSink, }; use crate::backfill::backup_sentinel::build_lsn_pair; use crate::backfill::backup_source::{BackupSink, BackupSource, EndInfo, StartInfo}; use crate::backfill::backup_source_direct::DirectSource; use crate::backfill::backup_source_object_store::ObjectStoreSource; -use crate::backfill::spool::{DEFERRED_SPOOL_MEM_MAX, DeferredSpool}; +use crate::backfill::spool::{DEFERRED_SPOOL_MEM_MAX, DeferredSpool, Replayable}; use crate::backfill::visibility_gate::{GateStats, resolve_phase, stream_phase}; use crate::backfill::visibility_pending::{self, PendingSpool}; use crate::backfill::wal_replay::{ @@ -68,7 +69,7 @@ use crate::backfill::wal_replay::{ }; use crate::backfill::walk_barrier::{WALK_CHECKPOINT_PERIOD, WalkBarrier}; use crate::decode::heap_decoder::{XLOG_HEAP_OPMASK, XLOG_HEAP_TRUNCATE}; -use crate::decode::visibility::{PgMultiXactAccum, PgXactAccum, PgXactPatch, PgXactView}; +use crate::decode::visibility::{PgXactPatch, SealedPatch}; use crate::decode::wal_xact::{ XLOG_XACT_ABORT, XLOG_XACT_ABORT_PREPARED, XLOG_XACT_COMMIT, XLOG_XACT_COMMIT_PREPARED, XLOG_XACT_OPMASK, parse_xact_payload, @@ -122,7 +123,7 @@ async fn run_base_backup_pass(ctx: &PassContext, reqs: &[BackupRequest]) -> Resu source, reqs, &tags, - PgXactPatch::new(), + SealedPatch::default(), None, &mut outcome, None, @@ -222,7 +223,7 @@ async fn run_object_store_pass(ctx: &PassContext, reqs: &[BackupRequest]) -> Res .context("backup_backfill: gap catalog pre-scan")?; (patch, segments) } else { - (PgXactPatch::new(), Vec::new()) + (SealedPatch::default(), Vec::new()) }; outcome.pg_xact_patch_len = patch.len(); @@ -280,7 +281,7 @@ async fn walk_and_ship( source: Box, reqs: &[BackupRequest], tags: &HashMap<(Oid, Oid), u64>, - patch: PgXactPatch, + patch: SealedPatch, replay: Option, outcome: &mut PassOutcome, backup_name: Option, @@ -354,7 +355,7 @@ async fn walk_and_ship( checkpoint.backup = backup.clone(); checkpoint.catalog = catalog; } - let (drain_outcome, pending) = if checkpoint.ready() { + let (drain_outcome, pending, resumed) = if checkpoint.ready() { outcome.counts = checkpoint.counts; let spool = DeferredSpool::resume( ctx.scratch_dir.join("bootstrap_deferred.bin"), @@ -368,15 +369,12 @@ async fn walk_and_ship( tracing::info!(target: "walshadow::backfill", offset = checkpoint.offset, bytes = checkpoint.spool.bytes, "resuming deferred backup replay"); ( - bootstrap::BootstrapDrainOutcome { - next_seq: 0, - rows_routed: 0, - deferred: Some(spool), - }, + bootstrap::BootstrapDrainOutcome::default(), PendingSpool::new(ctx.scratch_dir.join("gate_pending.bin"), filter.clone()), + Some(spool), ) } else { - run_walk( + let (drain_outcome, pending) = run_walk( ctx, source, &filter, @@ -388,7 +386,8 @@ async fn walk_and_ship( backup_name.is_some(), outcome, ) - .await? + .await?; + (drain_outcome, pending, None) }; let mut next_seq = drain_outcome.next_seq; let mut deferred = drain_outcome.deferred; @@ -413,13 +412,13 @@ async fn walk_and_ship( checkpoint.counts = outcome.counts; // Per-file resume is spent once the walk ends; drop it rather than // carry a path per relation segment through every later save - checkpoint.walk = WalkState { - done: true, - ..Default::default() - }; + checkpoint.walk = WalkState::finished(); checkpoint.save(&ctx.scratch_dir).await?; } let resumable = checkpoint.ready(); + let deferred = deferred + .map(Replayable::from) + .or(resumed.map(Replayable::from)); if let Some(spool) = deferred { let config = ctx.config_rx.as_ref().map(|rx| rx.borrow().clone()); let mapping = ctx.mapping.snapshot().await; @@ -514,7 +513,7 @@ async fn run_walk( source: Box, filter: &CatalogMap, lsn_overrides: HashMap<(Oid, Oid), u64>, - patch: PgXactPatch, + patch: SealedPatch, resolver: &ToastResolver, tail: &OwnedTail, checkpoint: &mut BackupCheckpoint, @@ -528,10 +527,7 @@ async fn run_walk( if !resuming { checkpoint.walk = Default::default(); } - let pg_xact = Arc::new(std::sync::Mutex::new(PgXactAccum::new())); - let pg_multixact = Arc::new(std::sync::Mutex::new(PgMultiXactAccum::new( - ctx.source_major, - ))); + let accums = GateAccums::new(ctx.source_major); let (walk_tx, walk_rx) = mpsc::channel::>(BOOTSTRAP_TUPLE_CHANNEL_CAP); let (gated_tx, gated_rx) = mpsc::channel::>(BOOTSTRAP_TUPLE_CHANNEL_CAP); @@ -547,10 +543,10 @@ async fn run_walk( let reopened = if resuming { tracing::info!( target: "walshadow::backfill", - files = checkpoint.walk.files.len(), - parts = checkpoint.walk.parts.len(), - gate_records = checkpoint.walk.gate_deferred.records, - toast_records = checkpoint.walk.toast_deferred.records, + files = checkpoint.walk.files().len(), + parts = checkpoint.walk.parts().len(), + gate_records = checkpoint.walk.gate_deferred().records, + toast_records = checkpoint.walk.toast_deferred().records, "resuming backup page walk", ); reopen_walk_spools(checkpoint, &gate_spool_path, &toast_spool_path) @@ -587,14 +583,13 @@ async fn run_walk( let barrier = resumable.then(|| Arc::new(WalkBarrier::default())); let mut walk_sink = PageWalkSink::new(filter.clone(), walk_tx, resolver.stores_chunks()) .with_stats(ctx.stats.backfill_backup_walk.clone()) - .with_pg_xact_accum(pg_xact.clone()) - .with_pg_multixact_accum(pg_multixact.clone()) + .with_gate(accums.clone()) .with_lsn_overrides(lsn_overrides); if let Some(b) = &barrier { walk_sink = walk_sink.with_resume( b.clone(), - checkpoint.walk.files.iter().cloned().collect(), - checkpoint.walk.parts.iter().cloned().collect(), + checkpoint.walk.files().iter().cloned().collect(), + checkpoint.walk.parts().iter().cloned().collect(), ); } let sink: Arc = Arc::new(walk_sink); @@ -606,8 +601,6 @@ async fn run_walk( walk_rx, gated_tx, filter.clone(), - pg_xact, - pg_multixact, patch, walk_ok_rx, gate_spool, @@ -639,7 +632,7 @@ async fn run_walk( => Err(e), }; if run_res.is_ok() { - let _ = walk_ok_tx.send(()); + let _ = walk_ok_tx.send(accums.complete()); } else { drop(walk_ok_tx); } @@ -689,10 +682,8 @@ async fn gate_task( mut rx: mpsc::Receiver>, tx: mpsc::Sender>, filter: CatalogMap, - pg_xact: Arc>, - pg_multixact: Arc>, - patch: PgXactPatch, - walk_ok: oneshot::Receiver<()>, + patch: SealedPatch, + walk_ok: oneshot::Receiver, mut deferred: DeferredSpool, mut pending: PendingSpool, barrier: Option>, @@ -708,17 +699,14 @@ async fn gate_task( barrier.as_ref(), ) .await?; - if walk_ok.await.is_err() { + let Ok(accums) = walk_ok.await else { stats.gated += stats.deferred; deferred.discard().await; // Resolution never ran; shipping reclaims the empty pending row spool return Ok((stats, 0, pending)); - } - // Take the accums out so no std guard is held across the sends below - let accum = std::mem::take(&mut *pg_xact.lock().expect("pg_xact accum lock")); - let multi = pg_multixact.lock().expect("pg_multixact accum lock").take(); - let segments = accum.segment_count(); - let view = PgXactView::new(&accum, &patch).with_multixact(&multi); + }; + let segments = accums.segment_count(); + let view = accums.view(&patch); resolve_phase(deferred, &view, &tx, Some(&mut pending), &mut stats).await?; Ok((stats, segments, pending)) } @@ -734,13 +722,13 @@ async fn reopen_walk_spools( DeferredSpool::reopen_at( gate_path.to_path_buf(), DEFERRED_SPOOL_MEM_MAX, - walk.gate_deferred, + walk.gate_deferred(), ) .await?, DeferredSpool::reopen_at( toast_path.to_path_buf(), DEFERRED_SPOOL_MEM_MAX, - walk.toast_deferred, + walk.toast_deferred(), ) .await?, )) @@ -763,17 +751,11 @@ async fn record_walked( let Some(proof) = barrier.collect().await else { continue; }; - if let Err(e) = tail.checkpoint(proof.next_seq).await { - return anyhow::Error::msg(e); - } - state.walk.files.extend(proof.files); - state.walk.gate_deferred = proof.gate_deferred; - state.walk.toast_deferred = proof.toast_deferred; - let recorded = state.walk.files.iter().cloned().collect(); - state - .walk - .parts - .extend(barrier.settled_parts(&recorded).await); + let walk = match proof.prove(tail).await { + Ok(walk) => walk, + Err(e) => return anyhow::Error::msg(e), + }; + state.walk.record(walk, barrier).await; if let Err(e) = state.save(dir).await { return e; } @@ -950,7 +932,7 @@ async fn prescan_gap( filter_oids: &HashSet, current_rfns: &HashMap, s_max: u64, -) -> Result { +) -> Result { let mut sink = PrescanSink { target_db_oid, filter_oids: filter_oids.clone(), @@ -967,8 +949,7 @@ async fn prescan_gap( fresher backup, or use initial_load='copy'" ); } - sink.patch.seal(); - Ok(sink.patch) + Ok(sink.patch.seal()) } /// `pg_class` / `pg_attribute` initial (mapped) filenodes; a rewrite of the @@ -1122,6 +1103,7 @@ mod tests { use super::*; use crate::decode::visibility::{ HEAP_XMAX_INVALID, HEAP_XMAX_IS_MULTI, HEAP_XMIN_COMMITTED, HEAP_XMIN_INVALID, + PgMultiXactAccum, PgXactAccum, PgXactView, }; use crate::decode::wal_xact::{XACT_XINFO_HAS_TWOPHASE, XLOG_XACT_HAS_INFO}; use crate::record::Route; @@ -1249,7 +1231,8 @@ mod tests { assert!(s.skew.is_none()); assert_eq!(s.patch.len(), 2); let accum = PgXactAccum::new(); - let view = PgXactView::new(&accum, &s.patch); + let patch = std::mem::take(&mut s.patch).seal(); + let view = PgXactView::new(&accum, &patch); assert_eq!( view.xid_status(700), crate::decode::visibility::XidStatus::Committed @@ -1288,7 +1271,8 @@ mod tests { )); assert!(s.skew.is_none()); let accum = PgXactAccum::new(); - let view = PgXactView::new(&accum, &s.patch); + let patch = std::mem::take(&mut s.patch).seal(); + let view = PgXactView::new(&accum, &patch); assert_eq!( view.xid_status(800), crate::decode::visibility::XidStatus::Committed @@ -1297,7 +1281,7 @@ mod tests { view.xid_status(801), crate::decode::visibility::XidStatus::Aborted ); - assert_eq!(s.patch.len(), 2, "finishing backend's xid stays unpatched"); + assert_eq!(patch.len(), 2, "finishing backend's xid stays unpatched"); } /// Reject truncated subxact payload @@ -1573,8 +1557,8 @@ mod tests { #[tokio::test] async fn gate_task_routes_hinted_defers_unhinted_and_resolves_at_eof() { let filter = CatalogMap::new(); - let pg_xact = Arc::new(std::sync::Mutex::new(PgXactAccum::new())); - let pg_multixact = Arc::new(std::sync::Mutex::new(PgMultiXactAccum::new(17))); + let pg_xact = PgXactAccum::new(); + let pg_multixact = PgMultiXactAccum::new(17); let mut patch = PgXactPatch::new(); patch.commit(500, &[]); patch.abort(600, &[]); @@ -1589,9 +1573,7 @@ mod tests { walk_rx, gated_tx, filter, - pg_xact, - pg_multixact, - patch, + patch.seal(), walk_ok_rx, DeferredSpool::new(spool_path.clone(), 0), test_pending(), @@ -1618,7 +1600,11 @@ mod tests { // Unhinted, gap-aborted writer: deferred, then gated via patch walk_tx.send(vec![tuple(16400, 600, 0, 0)]).await.unwrap(); drop(walk_tx); - walk_ok_tx.send(()).unwrap(); + assert!( + walk_ok_tx + .send(CompleteAccums::from_parts(pg_xact, pg_multixact)) + .is_ok() + ); let (stats, _segments, _pending) = gate.await.unwrap().unwrap(); let mut got = Vec::new(); @@ -1639,15 +1625,13 @@ mod tests { #[tokio::test] async fn gate_task_discards_deferred_without_walk_success() { let filter = CatalogMap::new(); - let pg_xact = Arc::new(std::sync::Mutex::new(PgXactAccum::new())); - let pg_multixact = Arc::new(std::sync::Mutex::new(PgMultiXactAccum::new(17))); let mut patch = PgXactPatch::new(); // Patch alone would emit xid 500; failure path must not consult it patch.commit(500, &[]); let (walk_tx, walk_rx) = mpsc::channel(16); let (gated_tx, mut gated_rx) = mpsc::channel(16); - let (walk_ok_tx, walk_ok_rx) = oneshot::channel::<()>(); + let (walk_ok_tx, walk_ok_rx) = oneshot::channel::(); // Threshold 0: discard must unlink the spool file let tmp = tempfile::tempdir().unwrap(); let spool_path = tmp.path().join("gate_deferred.bin"); @@ -1655,9 +1639,7 @@ mod tests { walk_rx, gated_tx, filter, - pg_xact, - pg_multixact, - patch, + patch.seal(), walk_ok_rx, DeferredSpool::new(spool_path.clone(), 0), test_pending(), @@ -1711,8 +1693,8 @@ mod tests { #[tokio::test] async fn gate_task_gates_multixact_with_committed_updater() { let filter = CatalogMap::new(); - let pg_xact = Arc::new(std::sync::Mutex::new(PgXactAccum::new())); - let pg_multixact = Arc::new(std::sync::Mutex::new(multi_with_updater_901())); + let pg_xact = PgXactAccum::new(); + let pg_multixact = multi_with_updater_901(); let mut patch = PgXactPatch::new(); patch.commit(901, &[]); @@ -1723,9 +1705,7 @@ mod tests { walk_rx, gated_tx, filter, - pg_xact, - pg_multixact, - patch, + patch.seal(), walk_ok_rx, mem_spool(), test_pending(), @@ -1742,7 +1722,11 @@ mod tests { .await .unwrap(); drop(walk_tx); - walk_ok_tx.send(()).unwrap(); + assert!( + walk_ok_tx + .send(CompleteAccums::from_parts(pg_xact, pg_multixact)) + .is_ok() + ); let (stats, _segments, _pending) = gate.await.unwrap().unwrap(); assert!(gated_rx.recv().await.is_none(), "dead tuple must not emit"); @@ -1754,9 +1738,9 @@ mod tests { #[tokio::test] async fn gate_task_errors_on_unresolvable_multixact() { let filter = CatalogMap::new(); - let pg_xact = Arc::new(std::sync::Mutex::new(PgXactAccum::new())); + let pg_xact = PgXactAccum::new(); // Empty accum: mxid below any collected segment ⇒ unresolvable - let pg_multixact = Arc::new(std::sync::Mutex::new(PgMultiXactAccum::new(17))); + let pg_multixact = PgMultiXactAccum::new(17); let (walk_tx, walk_rx) = mpsc::channel(16); let (gated_tx, _gated_rx) = mpsc::channel(16); @@ -1765,9 +1749,7 @@ mod tests { walk_rx, gated_tx, filter, - pg_xact, - pg_multixact, - PgXactPatch::new(), + SealedPatch::default(), walk_ok_rx, mem_spool(), test_pending(), @@ -1784,7 +1766,11 @@ mod tests { .await .unwrap(); drop(walk_tx); - walk_ok_tx.send(()).unwrap(); + assert!( + walk_ok_tx + .send(CompleteAccums::from_parts(pg_xact, pg_multixact)) + .is_ok() + ); let err = gate.await.unwrap().map(|_| ()).unwrap_err(); assert!(err.contains("pg_multixact"), "{err}"); diff --git a/src/backfill/backup_checkpoint.rs b/src/backfill/backup_checkpoint.rs index c52fe141..f4fa27d7 100644 --- a/src/backfill/backup_checkpoint.rs +++ b/src/backfill/backup_checkpoint.rs @@ -8,6 +8,7 @@ use serde::{Deserialize, Serialize}; use crate::backfill::backfill_staging::{StagingPlan, StagingSession}; use crate::backfill::backfill_types::{BackupRequest, WalkCounts}; use crate::backfill::spool::SpoolMark; +use crate::backfill::walk_barrier::{DurableWalk, WalkBarrier}; use crate::config::ResolvedConfig; use crate::emit::ch_emitter::EmitterConfig; use crate::mapping::MappingSnapshot; @@ -38,16 +39,52 @@ pub fn digest<'a>(parts: impl IntoIterator) -> u64 { /// produced is durable in staging or in one of the two spools below #[derive(Clone, Debug, Default, Serialize, Deserialize)] pub struct WalkState { - pub done: bool, + done: bool, /// Cluster-relative paths, eg `base/5/16400.2` - pub files: Vec, + files: Vec, /// Tar parts holding no SLRU file and no incomplete heap file, so a /// resumed walk need not fetch them at all - pub parts: Vec, + parts: Vec, /// Undecided-xid tuples the gate holds for walk EOF - pub gate_deferred: SpoolMark, + gate_deferred: SpoolMark, /// TOAST referrers the drain could not resolve yet - pub toast_deferred: SpoolMark, + toast_deferred: SpoolMark, +} + +impl WalkState { + /// Walk ended; per-file resume is spent + pub fn finished() -> Self { + Self { + done: true, + ..Self::default() + } + } + + pub fn files(&self) -> &[String] { + &self.files + } + + pub fn parts(&self) -> &[String] { + &self.parts + } + + pub fn gate_deferred(&self) -> SpoolMark { + self.gate_deferred + } + + pub fn toast_deferred(&self) -> SpoolMark { + self.toast_deferred + } + + /// Record a proven walk tick, then the parts it settled + pub async fn record(&mut self, walk: DurableWalk, barrier: &WalkBarrier) { + let (files, gate_deferred, toast_deferred) = walk.into_parts(); + self.files.extend(files); + self.gate_deferred = gate_deferred; + self.toast_deferred = toast_deferred; + let recorded = self.files.iter().cloned().collect(); + self.parts.extend(barrier.settled_parts(&recorded).await); + } } #[derive(Clone, Debug, Serialize, Deserialize)] diff --git a/src/backfill/backup_page_walk.rs b/src/backfill/backup_page_walk.rs index 94b6065d..42821bc6 100644 --- a/src/backfill/backup_page_walk.rs +++ b/src/backfill/backup_page_walk.rs @@ -32,6 +32,7 @@ use crate::backfill::walk_barrier::WalkBarrier; use crate::decode::heap_decoder::{ ColumnValue, CommittedTuple, DecodeError, DecodedHeap, DecodedTuple, HeapOp, decode_block_data, }; +use crate::decode::visibility::{PgMultiXactAccum, PgXactAccum, PgXactView, SealedPatch}; use crate::schema::RelDescriptor; use ahash::{HashMap, HashMapExt, HashSet}; @@ -461,12 +462,9 @@ pub struct PageWalkSink { /// `None` taps everything the catalog map holds. A superset filter, not a /// second source of truth: the drain's `skip_initial` stays the authority tap_filenodes: Option>>, - /// `pg_xact/` segments Tap into here for backfill visibility gate - /// (architecture/bootstrap.md); `None` (greenfield) keeps Skip. - pg_xact: Option>>, - /// `pg_multixact/{offsets,members}` segments, for multixact xmax - /// resolution in the same gate - pg_multixact: Option>>, + /// `pg_xact/` and `pg_multixact/` segments Tap into here for backfill + /// visibility gate (architecture/bootstrap.md); `None` (greenfield) keeps Skip + gate: Option, /// Resumable walk: heap files a checkpoint already proved durable, and the /// barrier every other file reports completion through resume: Option, @@ -507,18 +505,14 @@ impl PageWalkSink { captured: Arc::default(), store_toast, tap_filenodes: None, - pg_xact: None, - pg_multixact: None, + gate: None, resume: None, } } - /// Collect `pg_xact/` segments for the visibility gate. - pub fn with_pg_xact_accum( - mut self, - accum: Arc>, - ) -> Self { - self.pg_xact = Some(accum); + /// Collect `pg_xact/` and `pg_multixact/` segments for the visibility gate + pub fn with_gate(mut self, accums: GateAccums) -> Self { + self.gate = Some(accums); self } @@ -553,15 +547,6 @@ impl PageWalkSink { entry.1 |= slru; } - /// Collect `pg_multixact/` segments for multixact xmax resolution. - pub fn with_pg_multixact_accum( - mut self, - accum: Arc>, - ) -> Self { - self.pg_multixact = Some(accum); - self - } - /// Decline filenodes outside `set` at `begin`. Bytes still drain off the /// wire; tuples never decode pub fn with_tap_filenodes(mut self, set: Arc>) -> Self { @@ -594,8 +579,7 @@ impl PageWalkSink { captured: Arc::default(), store_toast: false, tap_filenodes: None, - pg_xact: None, - pg_multixact: None, + gate: None, resume: None, } } @@ -626,20 +610,15 @@ impl PageWalkSink { parse_base_path(&meta.path) } - fn classify_slru(&self, path: &std::path::Path) -> Option { - if self.pg_xact.is_some() - && let Some(segno) = crate::decode::visibility::pg_xact_segno_from_path(path) - { + fn classify_slru(path: &std::path::Path) -> Option { + use crate::decode::visibility::MultiXactSegment; + if let Some(segno) = crate::decode::visibility::pg_xact_segno_from_path(path) { return Some(SlruSegment::PgXact(segno)); } - if self.pg_multixact.is_some() { - use crate::decode::visibility::MultiXactSegment; - return match crate::decode::visibility::pg_multixact_segno_from_path(path)? { - MultiXactSegment::Offsets(s) => Some(SlruSegment::MultiOffsets(s)), - MultiXactSegment::Members(s) => Some(SlruSegment::MultiMembers(s)), - }; + match crate::decode::visibility::pg_multixact_segno_from_path(path)? { + MultiXactSegment::Offsets(s) => Some(SlruSegment::MultiOffsets(s)), + MultiXactSegment::Members(s) => Some(SlruSegment::MultiMembers(s)), } - None } } @@ -665,14 +644,14 @@ impl BackupSink for PageWalkSink { async fn begin(&self, meta: &FileMeta) -> io::Result { if matches!(meta.kind, FileKind::File) - && let Some(slru) = self.classify_slru(&meta.path) + && let Some(gate) = &self.gate + && let Some(slru) = Self::classify_slru(&meta.path) { self.tally(meta, None, true); return Ok(FileAction::Tap(Box::new(SlruEntry { seg: slru, buf: Vec::with_capacity(meta.size as usize), - pg_xact: self.pg_xact.clone(), - pg_multixact: self.pg_multixact.clone(), + gate: gate.clone(), }))); } let Some(f) = self.classify(meta) else { @@ -918,8 +897,59 @@ impl EntrySink for PageWalkEntry { struct SlruEntry { seg: SlruSegment, buf: Vec, - pg_xact: Option>>, - pg_multixact: Option>>, + gate: GateAccums, +} + +/// `pg_xact` and `pg_multixact` accums one gated walk fills +#[derive(Clone)] +pub struct GateAccums { + pg_xact: Arc>, + pg_multixact: Arc>, +} + +impl GateAccums { + pub fn new(source_major: u32) -> Self { + Self { + pg_xact: Arc::default(), + pg_multixact: Arc::new(std::sync::Mutex::new(PgMultiXactAccum::new(source_major))), + } + } + + /// Walk reached EOF without error, so every SLRU segment landed + pub fn complete(self) -> CompleteAccums { + CompleteAccums { + pg_xact: std::mem::take(&mut *self.pg_xact.lock().expect("pg_xact accum lock")), + pg_multixact: self + .pg_multixact + .lock() + .expect("pg_multixact accum lock") + .take(), + } + } +} + +/// Accums of a finished walk, the only ones deferred tuples resolve against +pub struct CompleteAccums { + pg_xact: PgXactAccum, + pg_multixact: PgMultiXactAccum, +} + +impl CompleteAccums { + #[cfg(test)] + pub(crate) fn from_parts(pg_xact: PgXactAccum, pg_multixact: PgMultiXactAccum) -> Self { + Self { + pg_xact, + pg_multixact, + } + } + + pub fn segment_count(&self) -> usize { + self.pg_xact.segment_count() + } + + pub fn view<'a>(&'a self, patch: &'a SealedPatch) -> PgXactView<'a> { + PgXactView::new(&self.pg_xact, patch).with_multixact(&self.pg_multixact) + } } #[async_trait] @@ -930,34 +960,16 @@ impl EntrySink for SlruEntry { } async fn end(self: Box) -> io::Result<()> { - let SlruEntry { - seg, - buf, - pg_xact, - pg_multixact, - } = *self; + let SlruEntry { seg, buf, gate } = *self; + let multi = || gate.pg_multixact.lock().expect("pg_multixact accum lock"); match seg { - SlruSegment::PgXact(segno) => { - if let Some(a) = pg_xact { - a.lock() - .expect("pg_xact accum lock") - .insert_segment(segno, buf); - } - } - SlruSegment::MultiOffsets(segno) => { - if let Some(a) = pg_multixact { - a.lock() - .expect("pg_multixact accum lock") - .insert_offsets_segment(segno, buf); - } - } - SlruSegment::MultiMembers(segno) => { - if let Some(a) = pg_multixact { - a.lock() - .expect("pg_multixact accum lock") - .insert_members_segment(segno, buf); - } - } + SlruSegment::PgXact(segno) => gate + .pg_xact + .lock() + .expect("pg_xact accum lock") + .insert_segment(segno, buf), + SlruSegment::MultiOffsets(segno) => multi().insert_offsets_segment(segno, buf), + SlruSegment::MultiMembers(segno) => multi().insert_members_segment(segno, buf), } Ok(()) } diff --git a/src/backfill/bootstrap_marker.rs b/src/backfill/bootstrap_marker.rs index 20936233..3c424531 100644 --- a/src/backfill/bootstrap_marker.rs +++ b/src/backfill/bootstrap_marker.rs @@ -4,6 +4,7 @@ use std::path::{Path, PathBuf}; use anyhow::{Context, Result}; use crate::ch_emitter::BootstrapMode; +use crate::visibility_pending::PendingRecorded; pub const MARKER_FILENAME: &str = "walshadow_bootstrap.incomplete"; @@ -66,7 +67,7 @@ impl ExtractedCheckpoint { .context("persist extracted checkpoint") } - pub async fn clear(dir: &Path) -> Result<()> { + async fn clear(dir: &Path) -> Result<()> { match tokio::fs::remove_file(dir.join(EXTRACTED_FILENAME)).await { Ok(()) => {} Err(e) if e.kind() == io::ErrorKind::NotFound => return Ok(()), @@ -108,7 +109,14 @@ impl BootstrapMarker { .context("persist bootstrap marker") } - pub async fn clear(dir: &Path) -> Result<()> { + /// End the attempt. Pending rows already in ClickHouse must be recorded + /// first, so a restart can still publish them + pub async fn complete(dir: &Path, _: PendingRecorded) -> Result<()> { + ExtractedCheckpoint::clear(dir).await?; + Self::clear(dir).await + } + + async fn clear(dir: &Path) -> Result<()> { tokio::fs::remove_file(dir.join(MARKER_FILENAME)) .await .context("clear completed bootstrap marker")?; diff --git a/src/backfill/bootstrap_window.rs b/src/backfill/bootstrap_window.rs index 7a9310af..56c5642a 100644 --- a/src/backfill/bootstrap_window.rs +++ b/src/backfill/bootstrap_window.rs @@ -322,9 +322,10 @@ async fn run_live( feed.start_physical_replication(None, begin, timeline) .await .context("bootstrap window leg: START_REPLICATION")?; - let mut stream = WalStream::new(timeline, WAL_SEG_SIZE, Pos::new(begin)) + let mut stream = WalStream::builder(timeline, WAL_SEG_SIZE, Pos::new(begin)) .map_err(|e| anyhow::anyhow!("bootstrap window leg: WalStream: {e}"))?; stream.filter_mut().set_target_db(leg.db_oid); + let mut stream = stream.start(); let mut seg_sink = DropSegments; let mut buf = Vec::new(); diff --git a/src/backfill/copy_backfill.rs b/src/backfill/copy_backfill.rs index e876bb52..20bfecb5 100644 --- a/src/backfill/copy_backfill.rs +++ b/src/backfill/copy_backfill.rs @@ -176,18 +176,44 @@ fn default_ledger_mode() -> String { } /// One backfill's durable state; boot re-runs `mode` at `s_lsn` while -/// `!done`, or resumes the swap tail while `swapped`. +/// pending, or resumes the swap tail while swapped. #[derive(Debug, Clone)] struct LedgerRec { s_lsn: Pos, - done: bool, mode: InitialLoadMode, - swapped: bool, - staging_uuid: Option, - copy: Option, + phase: Phase, toast_seeded: bool, } +#[derive(Debug, Clone, PartialEq, Eq)] +enum Phase { + Pending { + copy: Option, + }, + /// Staging uuid recorded before the exchange + Swapped { + staging_uuid: String, + }, + Done, +} + +impl LedgerRec { + fn done(&self) -> bool { + self.phase == Phase::Done + } +} + +/// Ledger durably records the swap for `rel`, so EXCHANGE may run +pub struct SwapPermit<'a> { + rel: &'a StagingRel, +} + +impl<'a> SwapPermit<'a> { + pub fn rel(&self) -> &'a StagingRel { + self.rel + } +} + /// Where a chunked COPY stopped. Every chunk below `next_block` proved its /// rows durable before this was written, so resume replays at most the chunk /// in flight when the daemon died @@ -240,25 +266,31 @@ impl Ledger { // Only this daemon writes modes; an unparseable one // degrades to re-COPY (idempotent) let mode = e.mode.parse().unwrap_or(InitialLoadMode::Copy); - ( - RelName::new(&e.namespace, &e.relname), - LedgerRec { - s_lsn: e.s_lsn, - done: e.done, - mode, - swapped: e.swapped, - staging_uuid: e.staging_uuid, + let rel = RelName::new(&e.namespace, &e.relname); + let phase = match (e.done, e.swapped, e.staging_uuid) { + (true, ..) => Phase::Done, + (false, true, Some(staging_uuid)) => Phase::Swapped { staging_uuid }, + (false, true, None) => { + return Err(invalid(format!("{rel} swapped without staging uuid"))); + } + (false, false, _) => Phase::Pending { copy: e.copy_relfilenode.zip(e.copy_next_block).map( |(relfilenode, next_block)| CopyCursor { relfilenode, next_block, }, ), - toast_seeded: e.toast_seeded, }, - ) + }; + let rec = LedgerRec { + s_lsn: e.s_lsn, + mode, + phase, + toast_seeded: e.toast_seeded, + }; + Ok((rel, rec)) }) - .collect(); + .collect::>()?; if unstamped { ledger.persist().await?; } @@ -273,17 +305,24 @@ impl Ledger { backfill: self .entries .iter() - .map(|(rel, rec)| LedgerEntry { - namespace: rel.namespace.to_string(), - relname: rel.name.to_string(), - s_lsn: rec.s_lsn, - done: rec.done, - mode: rec.mode.as_str().into(), - swapped: rec.swapped, - staging_uuid: rec.staging_uuid.clone(), - copy_relfilenode: rec.copy.map(|c| c.relfilenode), - copy_next_block: rec.copy.map(|c| c.next_block), - toast_seeded: rec.toast_seeded, + .map(|(rel, rec)| { + let (staging_uuid, copy) = match &rec.phase { + Phase::Pending { copy } => (None, *copy), + Phase::Swapped { staging_uuid } => (Some(staging_uuid.clone()), None), + Phase::Done => (None, None), + }; + LedgerEntry { + namespace: rel.namespace.to_string(), + relname: rel.name.to_string(), + s_lsn: rec.s_lsn, + done: rec.done(), + mode: rec.mode.as_str().into(), + swapped: staging_uuid.is_some(), + staging_uuid, + copy_relfilenode: copy.map(|c| c.relfilenode), + copy_next_block: copy.map(|c| c.next_block), + toast_seeded: rec.toast_seeded, + } }) .collect(), }; @@ -300,9 +339,7 @@ impl Ledger { let Some(rec) = self.entries.get_mut(rel) else { return Ok(false); }; - if rec.done - || rec.swapped - || rec.staging_uuid.is_some() + if !matches!(rec.phase, Phase::Pending { .. }) || rec.mode != mode || rec.s_lsn.get() != s_lsn { @@ -319,21 +356,25 @@ impl Ledger { /// Advance the COPY cursor. Callers write it only once the chunk below /// `next_block` proved durable async fn note_copy(&mut self, rel: &RelName, cursor: CopyCursor) -> std::io::Result<()> { - let Some(rec) = self.entries.get_mut(rel) else { + let Some(LedgerRec { + phase: Phase::Pending { copy }, + .. + }) = self.entries.get_mut(rel) + else { return Ok(()); }; - rec.copy = Some(cursor); + *copy = Some(cursor); self.persist().await } fn pending_count(&self) -> u64 { - self.entries.values().filter(|r| !r.done).count() as u64 + self.entries.values().filter(|r| !r.done()).count() as u64 } fn pending_count_for(&self, mode: InitialLoadMode) -> u64 { self.entries .values() - .filter(|r| !r.done && r.mode == mode) + .filter(|r| !r.done() && r.mode == mode) .count() as u64 } } @@ -753,6 +794,18 @@ impl CopyBackfiller { pending_rows: SharedPendingLedger, ) -> std::io::Result { let ledger = Ledger::load(spill_dir, system_id).await?; + { + let mut rows = pending_rows.lock().await; + let swapped = |r: &RelName| { + ledger + .entries + .get(r) + .is_some_and(|rec| matches!(rec.phase, Phase::Swapped { .. })) + }; + if rows.discard_held_unless(swapped) { + rows.persist().await?; + } + } let pending = AtomicU64::new(ledger.pending_count()); let pending_by_mode = [ AtomicU64::new(ledger.pending_count_for(InitialLoadMode::Copy)), @@ -855,7 +908,7 @@ impl CopyBackfiller { let (s_lsn, mode, spawn_pass, resume) = { let mut inner = self.inner.lock().await; let (s_lsn, mode) = match inner.ledger.entries.get(&rel) { - Some(rec) if rec.done => return, + Some(rec) if rec.done() => return, // Boot re-runs the recorded mode at the recorded S; the // config row's current mode applies only to a fresh entry Some(rec) => (rec.s_lsn, rec.mode), @@ -864,11 +917,8 @@ impl CopyBackfiller { rel.clone(), LedgerRec { s_lsn: Pos::new(opt_in_lsn), - done: false, mode, - swapped: false, - staging_uuid: None, - copy: None, + phase: Phase::Pending { copy: None }, toast_seeded: false, }, ); @@ -890,12 +940,12 @@ impl CopyBackfiller { // Swapped entry: pass rows already durable, exchange issued or // withheld — resume the swap tail, never re-load (the staging // name may hold the only copy of the live-window rows) - let resume = inner - .ledger - .entries - .get(&rel) - .filter(|r| r.swapped) - .cloned(); + let resume = inner.ledger.entries.get(&rel).and_then(|r| { + let Phase::Swapped { staging_uuid } = &r.phase else { + return None; + }; + Some((r.s_lsn, staging_uuid.clone())) + }); let mut spawn_pass = false; if resume.is_none() && matches!( @@ -913,9 +963,9 @@ impl CopyBackfiller { } (s_lsn, mode, spawn_pass, resume) }; - if let Some(rec) = resume { + if let Some((s_lsn, staging_uuid)) = resume { let this = self.clone(); - tokio::spawn(async move { this.resume_swap(rel, rec).await }); + tokio::spawn(async move { this.resume_swap(rel, s_lsn, staging_uuid).await }); return; } match mode { @@ -1136,8 +1186,9 @@ impl CopyBackfiller { checkpoint, }; let outcome = crate::backfill::backup_backfill::run_pass(&ctx, mode, reqs).await?; + self.hold_pending(&outcome).await; self.publish_staged(&staging, reqs).await; - self.record_pending(&outcome).await; + self.release_pending(&outcome).await; BackupCheckpoint::discard(&ctx.scratch_dir).await?; tokio::fs::remove_file(ctx.scratch_dir.join("bootstrap_deferred.bin")) .await @@ -1145,20 +1196,66 @@ impl CopyBackfiller { Ok(outcome) } - /// Record after publication so EXCHANGE cannot discard promoted rows. - /// Persist retries rather than failing: the backfill entry is already - /// done, so this ledger alone names the pending tables, and every live - /// settle persists the shared instance too - async fn record_pending(&self, outcome: &PassOutcome) { + /// Record before EXCHANGE so a swap failing past `mark_swapped` leaves + /// manifests for swap resume, held so promotion cannot precede the swap + async fn hold_pending(&self, outcome: &PassOutcome) { if outcome.pending_tables.is_empty() { return; } { let mut ledger = self.pending_rows.lock().await; for m in &outcome.pending_tables { - ledger.stage(m); + ledger.hold(m); } } + self.persist_pending().await; + } + + /// Publish activated what it exchanged; keep entries a swapped rel + /// resumes, discard the rest so their retry rebuilds pending tables + async fn release_pending(&self, outcome: &PassOutcome) { + if outcome.pending_tables.is_empty() { + return; + } + let unswapped: Vec<&RelName> = { + let inner = self.inner.lock().await; + outcome + .pending_tables + .iter() + .map(|m| &m.rel.rel) + .filter(|r| { + !inner + .ledger + .entries + .get(*r) + .is_some_and(|rec| matches!(rec.phase, Phase::Swapped { .. })) + }) + .collect() + }; + let mut discarded = false; + { + let mut ledger = self.pending_rows.lock().await; + for rel in unswapped { + discarded |= ledger.discard_held(rel); + } + } + if discarded { + self.persist_pending().await; + } + self.settle_pending().await; + } + + /// Persist before the done mark, else a crash leaves rows held forever + async fn activate_pending(&self, name: &RelName) { + if self.pending_rows.lock().await.activate(name) { + self.persist_pending().await; + } + } + + /// Retry rather than fail: the backfill entry may already be done, so + /// this ledger alone names the pending tables, and every live settle + /// persists the shared instance too + async fn persist_pending(&self) { let mut backoff = Duration::from_millis(100); while let Err(e) = self.pending_rows.lock().await.persist().await { tracing::error!( @@ -1170,6 +1267,9 @@ impl CopyBackfiller { tokio::time::sleep(backoff).await; backoff = (backoff * 2).min(Duration::from_secs(30)); } + } + + async fn settle_pending(&self) { if let Err(e) = self.settle_ended_pending().await { tracing::error!( target: "walshadow::backfill", @@ -1181,26 +1281,25 @@ impl CopyBackfiller { } /// Live apply folds only outcomes arriving after an entry exists. Xids - /// ending between the pass's replay cut and [`Self::record_pending`] - /// went by already, so source `pg_xact` decides them + /// ending between the pass's replay cut and [`Self::release_pending`] + /// went by already, so source `pg_xact` decides them. Settles even + /// without new outcomes: activated entries carry outcomes noted while held async fn settle_ended_pending(&self) -> anyhow::Result<()> { - let xids: Vec = self - .pending_rows - .lock() - .await - .outstanding() - .into_iter() - .collect(); - if xids.is_empty() { - return Ok(()); - } - let client = open_sql_client(&self.source_pg()) - .await - .context("pending visibility: source sql connect")?; - let ended = xid_outcomes(&client, &xids).await?; - if ended.is_empty() { - return Ok(()); - } + let xids: Vec = { + let ledger = self.pending_rows.lock().await; + if ledger.is_empty() { + return Ok(()); + } + ledger.outstanding().into_iter().collect() + }; + let ended = if xids.is_empty() { + Vec::new() + } else { + let client = open_sql_client(&self.source_pg()) + .await + .context("pending visibility: source sql connect")?; + xid_outcomes(&client, &xids).await? + }; // Connect before locking: live apply folds every commit under it let mut sess = StagingSession::connect(self.dest.clone()).await?; let mut ledger = self.pending_rows.lock().await; @@ -1208,9 +1307,6 @@ impl CopyBackfiller { for (xid, committed) in ended { settled += ledger.note(xid, &[], committed); } - if settled == 0 { - return Ok(()); - } self.stats .pending_xacts_settled .fetch_add(settled, Ordering::Relaxed); @@ -1229,6 +1325,7 @@ impl CopyBackfiller { let staged: HashSet<&RelName> = plan.rels.iter().map(|r| &r.rel).collect(); for r in reqs { if !staged.contains(&r.desc.rel_name) { + self.activate_pending(&r.desc.rel_name).await; self.mark_done_entry(&r.desc.rel_name).await; } } @@ -1318,12 +1415,13 @@ impl CopyBackfiller { // Persist precedes EXCHANGE: post-swap the staging name holds the // only copy of the live-window rows, and a pending-looking entry // would re-run the pass and rebuild staging over it - if !self.mark_swapped(&rel.rel, &uuid).await { + let Some(permit) = self.mark_swapped(rel, uuid).await else { anyhow::bail!("ledger persist failed; exchange withheld"); - } + }; crate::ops::stages::PUBLISH - .measure(sess.exchange(rel)) + .measure(sess.exchange(permit)) .await?; + self.activate_pending(&rel.rel).await; Ok(true) } @@ -1344,8 +1442,13 @@ impl CopyBackfiller { /// phase apart: unchanged = exchange never applied (staging still holds /// the load), changed = exchange applied (staging holds the pre-swap /// storage), missing = copy-back + drop ran, only the done mark is owed. - async fn resume_swap(self: Arc, name: RelName, rec: LedgerRec) { - if let Err(e) = self.resume_swap_inner(&name, &rec).await { + async fn resume_swap( + self: Arc, + name: RelName, + s_lsn: Pos, + staging_uuid: String, + ) { + if let Err(e) = self.resume_swap_inner(&name, s_lsn, &staging_uuid).await { tracing::error!( target: "walshadow::backfill", qname = %name, @@ -1358,7 +1461,12 @@ impl CopyBackfiller { self.refresh_gauges(&inner.ledger); } - async fn resume_swap_inner(&self, name: &RelName, rec: &LedgerRec) -> anyhow::Result<()> { + async fn resume_swap_inner( + &self, + name: &RelName, + s_lsn: Pos, + staging_uuid: &str, + ) -> anyhow::Result<()> { let target = self .mapping .with(|m| m.get(name).map(|t| t.target.clone())) @@ -1368,17 +1476,19 @@ impl CopyBackfiller { rel: name.clone(), database: target.database, table: target.table, - s_lsn: rec.s_lsn.get(), + s_lsn: s_lsn.get(), }; let mut sess = StagingSession::connect(self.dest.clone()) .await? .with_rules(self.table_rules()); match sess.table_uuid(&rel.database, &rel.staging_table()).await? { None => { + self.activate_pending(name).await; self.mark_done_entry(name).await; + self.settle_pending().await; return Ok(()); } - Some(u) if Some(&u) == rec.staging_uuid.as_ref() => { + Some(u) if u == staging_uuid => { // Schema may have moved while down — same gate as the pass let real_fp = sess.schema_fingerprint(&rel.database, &rel.table).await?; let staging_fp = sess @@ -1386,31 +1496,35 @@ impl CopyBackfiller { .await?; if real_fp != staging_fp { sess.drop_staging(&rel).await?; + if self.pending_rows.lock().await.discard_held(name) { + self.persist_pending().await; + } self.clear_swapped(name).await; anyhow::bail!( "destination schema changed before exchange; load discarded, entry re-pends" ); } - sess.exchange(&rel).await?; + // Ledger recorded this uuid as swapped before the crash + sess.exchange(SwapPermit { rel: &rel }).await?; } Some(_) => {} } + self.activate_pending(name).await; tokio::time::sleep(self.dest.current().insert_timeout).await; sess.copy_back(&rel).await?; sess.drop_staging(&rel).await?; self.mark_done_entry(name).await; + self.settle_pending().await; Ok(()) } - /// `false` (caller must not exchange) when the entry vanished (opt-out - /// raced the publish) or the persist failed. - async fn mark_swapped(&self, name: &RelName, uuid: &str) -> bool { + /// `None` when the entry vanished (opt-out raced the publish) or the + /// persist failed. + async fn mark_swapped<'a>(&self, rel: &'a StagingRel, uuid: String) -> Option> { + let name = &rel.rel; let mut inner = self.inner.lock().await; - let Some(rec) = inner.ledger.entries.get_mut(name) else { - return false; - }; - rec.swapped = true; - rec.staging_uuid = Some(uuid.to_owned()); + let rec = inner.ledger.entries.get_mut(name)?; + let prior = std::mem::replace(&mut rec.phase, Phase::Swapped { staging_uuid: uuid }); if let Err(e) = inner.ledger.persist().await { tracing::warn!( target: "walshadow::backfill", @@ -1418,21 +1532,21 @@ impl CopyBackfiller { error = %e, "ledger persist failed; exchange withheld, entry stays pending", ); - let rec = inner.ledger.entries.get_mut(name).expect("just present"); - rec.swapped = false; - rec.staging_uuid = None; - return false; + inner + .ledger + .entries + .get_mut(name) + .expect("just present") + .phase = prior; + return None; } - true + Some(SwapPermit { rel }) } async fn mark_done_entry(&self, name: &RelName) { let mut inner = self.inner.lock().await; if let Some(rec) = inner.ledger.entries.get_mut(name) { - rec.done = true; - rec.swapped = false; - rec.staging_uuid = None; - rec.copy = None; + rec.phase = Phase::Done; if let Err(e) = inner.ledger.persist().await { tracing::warn!( target: "walshadow::backfill", @@ -1448,8 +1562,7 @@ impl CopyBackfiller { async fn clear_swapped(&self, name: &RelName) { let mut inner = self.inner.lock().await; if let Some(rec) = inner.ledger.entries.get_mut(name) { - rec.swapped = false; - rec.staging_uuid = None; + rec.phase = Phase::Pending { copy: None }; if let Err(e) = inner.ledger.persist().await { tracing::warn!( target: "walshadow::backfill", @@ -1482,8 +1595,7 @@ impl CopyBackfiller { if let Some(entry) = inner.ledger.entries.get_mut(&desc.rel_name) && entry.s_lsn == s_lsn { - entry.done = true; - entry.copy = None; + entry.phase = Phase::Done; if let Err(e) = inner.ledger.persist().await { tracing::warn!( target: "walshadow::backfill", @@ -1754,7 +1866,11 @@ impl CopyBackfiller { } async fn copy_cursor(&self, rel: &RelName) -> Option { - self.inner.lock().await.ledger.entries.get(rel)?.copy + let inner = self.inner.lock().await; + let Phase::Pending { copy } = inner.ledger.entries.get(rel)?.phase else { + return None; + }; + copy } /// Cursor loss only costs a repeated chunk, so a failed persist logs @@ -1980,25 +2096,20 @@ mod tests { let mut ledger = Ledger::load(tmp.path(), 7).await.unwrap(); let pending = LedgerRec { s_lsn: 100.into(), - done: false, mode, - swapped: false, - staging_uuid: None, - copy: None, + phase: Phase::Pending { copy: None }, toast_seeded: false, }; assert!(!ledger.fallback_to_copy(&rel, mode, 100).await.unwrap()); for rec in [ LedgerRec { - done: true, - ..pending.clone() - }, - LedgerRec { - swapped: true, + phase: Phase::Done, ..pending.clone() }, LedgerRec { - staging_uuid: Some("uuid".into()), + phase: Phase::Swapped { + staging_uuid: "uuid".into(), + }, ..pending.clone() }, LedgerRec { @@ -2030,7 +2141,7 @@ mod tests { let resumed = Ledger::load(tmp.path(), 7).await.unwrap(); assert_eq!(resumed.entries[&rel].mode, InitialLoadMode::Copy); assert_eq!(resumed.entries[&rel].s_lsn.get(), 100); - assert!(!resumed.entries[&rel].done); + assert!(!resumed.entries[&rel].done()); } #[test] @@ -2054,14 +2165,13 @@ mod tests { RelName::new("app", "orders"), LedgerRec { s_lsn: 0x1000.into(), - done: false, mode: InitialLoadMode::Copy, - swapped: false, - staging_uuid: None, - copy: Some(CopyCursor { - relfilenode: 16400, - next_block: 512, - }), + phase: Phase::Pending { + copy: Some(CopyCursor { + relfilenode: 16400, + next_block: 512, + }), + }, toast_seeded: true, }, ); @@ -2069,11 +2179,8 @@ mod tests { RelName::new("app", "done"), LedgerRec { s_lsn: 0x800.into(), - done: true, mode: InitialLoadMode::ObjectStore, - swapped: false, - staging_uuid: None, - copy: None, + phase: Phase::Done, toast_seeded: false, }, ); @@ -2081,11 +2188,10 @@ mod tests { RelName::new("app", "mid_swap"), LedgerRec { s_lsn: 0x2000.into(), - done: false, mode: InitialLoadMode::ObjectStore, - swapped: true, - staging_uuid: Some("a-uuid".into()), - copy: None, + phase: Phase::Swapped { + staging_uuid: "a-uuid".into(), + }, toast_seeded: false, }, ); @@ -2093,24 +2199,29 @@ mod tests { let again = Ledger::load(tmp.path(), 7).await.unwrap(); let orders = again.entries.get(&RelName::new("app", "orders")).unwrap(); - assert_eq!((orders.s_lsn.get(), orders.done), (0x1000, false)); + assert_eq!(orders.s_lsn.get(), 0x1000); assert_eq!(orders.mode, InitialLoadMode::Copy); - assert!(!orders.swapped); assert_eq!( - orders.copy, - Some(CopyCursor { - relfilenode: 16400, - next_block: 512 - }), + orders.phase, + Phase::Pending { + copy: Some(CopyCursor { + relfilenode: 16400, + next_block: 512 + }) + }, "COPY cursor round-trips", ); let done = again.entries.get(&RelName::new("app", "done")).unwrap(); - assert_eq!(done.copy, None); - assert_eq!((done.s_lsn.get(), done.done), (0x800, true)); + assert_eq!((done.s_lsn.get(), &done.phase), (0x800, &Phase::Done)); assert_eq!(done.mode, InitialLoadMode::ObjectStore, "mode round-trips"); let mid = again.entries.get(&RelName::new("app", "mid_swap")).unwrap(); - assert!(mid.swapped, "swap phase round-trips"); - assert_eq!(mid.staging_uuid.as_deref(), Some("a-uuid")); + assert_eq!( + mid.phase, + Phase::Swapped { + staging_uuid: "a-uuid".into() + }, + "swap phase round-trips" + ); assert_eq!(again.pending_count(), 2, "swapped counts as pending"); assert_eq!(again.pending_count_for(InitialLoadMode::Copy), 1); assert_eq!(again.pending_count_for(InitialLoadMode::ObjectStore), 1); @@ -2123,6 +2234,21 @@ mod tests { "{foreign}" ); + tokio::fs::write( + tmp.path().join(LEDGER_FILENAME), + "version = 1\nsystem_id = 7\n[[backfill]]\nnamespace = 'app'\nrelname = 'x'\n\ + s_lsn = '0/64'\ndone = false\nswapped = true\n", + ) + .await + .unwrap(); + let unpaired = Ledger::load(tmp.path(), 7).await.err().unwrap(); + assert!( + unpaired + .to_string() + .contains("swapped without staging uuid"), + "{unpaired}" + ); + tokio::fs::write(tmp.path().join(LEDGER_FILENAME), b"not json") .await .unwrap(); diff --git a/src/backfill/spool.rs b/src/backfill/spool.rs index 44f5a145..13aee188 100644 --- a/src/backfill/spool.rs +++ b/src/backfill/spool.rs @@ -39,16 +39,37 @@ type Result = std::result::Result; /// Insertion order preserved; once the file exists every record (prefix /// included) lives there. pub struct DeferredSpool { - mem: Vec, - mem_bytes: usize, + store: Store, mem_max: usize, path: PathBuf, - out: Option, + records: u64, + spooled_bytes: u64, +} + +enum Store { + Mem { + tuples: Vec, + bytes: usize, + }, + File(ChunkWriter), +} + +/// Checkpointed spool reopened read-only at an acknowledged record boundary +#[derive(Debug)] +pub struct ResumedSpool { + path: PathBuf, records: u64, spooled_bytes: u64, read_offset: u64, } +/// Deferred records ready to replay +#[derive(Debug)] +pub enum Replayable { + Open(DeferredSpool), + Resumed(ResumedSpool), +} + /// Durable length of an append-only spool. Records pair with bytes so /// [`DeferredSpool::reopen_at`] can refuse a file a crash left shorter than /// whoever recorded the mark counted @@ -74,14 +95,24 @@ impl DeferredSpool { /// be creatable pub fn new(path: PathBuf, mem_max: usize) -> Self { Self { - mem: Vec::new(), - mem_bytes: 0, + store: Store::Mem { + tuples: Vec::new(), + bytes: 0, + }, mem_max, path, - out: None, records: 0, spooled_bytes: 0, - read_offset: 0, + } + } + + fn at_file(path: PathBuf, mem_max: usize, out: ChunkWriter, mark: SpoolMark) -> Self { + Self { + store: Store::File(out), + mem_max, + path, + records: mark.records, + spooled_bytes: mark.bytes, } } @@ -95,7 +126,10 @@ impl DeferredSpool { /// Bytes retained in the in-memory prefix pub fn resident_bytes(&self) -> usize { - self.mem_bytes + match &self.store { + Store::Mem { bytes, .. } => *bytes, + Store::File(_) => 0, + } } /// Encoded bytes written to the spool file @@ -105,66 +139,56 @@ impl DeferredSpool { pub async fn push(&mut self, value: BackfillTuple) -> Result<()> { self.records += 1; - if self.out.is_none() { + if let Store::Mem { tuples, bytes } = &mut self.store { let value_bytes = approx_bytes(&value); - if self.mem_bytes + value_bytes <= self.mem_max { - self.mem_bytes += value_bytes; - self.mem.push(value); + if *bytes + value_bytes <= self.mem_max { + *bytes += value_bytes; + tuples.push(value); return Ok(()); } - self.create_and_flush_prefix().await?; - } - self.append(&value).await - } - - async fn create_and_flush_prefix(&mut self) -> Result<()> { - let parent = self.path.parent().filter(|p| !p.as_os_str().is_empty()); - if let Some(parent) = parent { - tokio::fs::create_dir_all(parent).await?; - } - let file = OpenOptions::new() - .write(true) - .create_new(true) - .open(&self.path) - .await?; - // Checkpoint names this file, so its directory entry must survive - crate::fs::fsync_dir(parent.unwrap_or(std::path::Path::new("."))).await?; - let mut out = ChunkWriter::new(file.into_std().await, 0); - out.buf.extend_from_slice(&SPOOL_MAGIC); - push_u16(&mut out.buf, SPOOL_VERSION); - self.out = Some(out); - for v in std::mem::take(&mut self.mem) { - self.append(&v).await?; } - self.mem_bytes = 0; + let out = self.file().await?; + self.spooled_bytes += append(out, &value).await?; Ok(()) } - async fn append(&mut self, value: &BackfillTuple) -> Result<()> { - let out = self.out.as_mut().expect("append without file"); - let buf = &mut out.buf; - let len_at = buf.len(); - push_u32(buf, 0); - let body_at = buf.len(); - encode_record(value, buf); - let len = (buf.len() - body_at) as u32; - buf[len_at..body_at].copy_from_slice(&len.to_le_bytes()); - self.spooled_bytes += 4 + u64::from(len); - out.maybe_flush().await + /// Writer, creating the file and moving the prefix into it on first use + async fn file(&mut self) -> Result<&mut ChunkWriter> { + if let Store::Mem { tuples, .. } = &mut self.store { + let tuples = std::mem::take(tuples); + let parent = self.path.parent().filter(|p| !p.as_os_str().is_empty()); + if let Some(parent) = parent { + tokio::fs::create_dir_all(parent).await?; + } + let file = OpenOptions::new() + .write(true) + .create_new(true) + .open(&self.path) + .await?; + // Checkpoint names this file, so its directory entry must survive + crate::fs::fsync_dir(parent.unwrap_or(std::path::Path::new("."))).await?; + let mut out = ChunkWriter::new(file.into_std().await, 0); + out.buf.extend_from_slice(&SPOOL_MAGIC); + push_u16(&mut out.buf, SPOOL_VERSION); + for v in &tuples { + self.spooled_bytes += append(&mut out, v).await?; + } + self.store = Store::File(out); + } + let Store::File(out) = &mut self.store else { + unreachable!("file store installed above"); + }; + Ok(out) } /// Unlike the xact spill's disposable contract, a bootstrap gate spool /// outlives a crash: the tuples in it came from backup pages nothing can /// re-read, so a resumed load replays them instead of the whole backup pub async fn checkpoint(&mut self) -> Result { - if self.out.is_none() { - if self.mem.is_empty() { - return Ok(SpoolMark::default()); - } - self.create_and_flush_prefix().await?; + if matches!(&self.store, Store::Mem { tuples, .. } if tuples.is_empty()) { + return Ok(SpoolMark::default()); } - let out = self.out.as_mut().expect("file after prefix flush"); - out.sync_data().await?; + self.file().await?.sync_data().await?; Ok(SpoolMark { records: self.records, bytes: self.spooled_bytes, @@ -191,12 +215,12 @@ impl DeferredSpool { let mut file = OpenOptions::new().write(true).open(&path).await?; file.set_len(valid_len).await?; file.seek(SeekFrom::Start(valid_len)).await?; - Ok(Self { - out: Some(ChunkWriter::new(file.into_std().await, valid_len)), + let out = ChunkWriter::new(file.into_std().await, valid_len); + let mark = SpoolMark { records, - spooled_bytes: valid_len.saturating_sub(4), - ..Self::new(path, mem_max) - }) + bytes: valid_len.saturating_sub(4), + }; + Ok(Self::at_file(path, mem_max, out, mark)) } /// Reopen for appends at a checkpointed length, discarding whatever the @@ -223,16 +247,12 @@ impl DeferredSpool { let mut file = OpenOptions::new().write(true).open(&path).await?; file.set_len(keep).await?; file.seek(SeekFrom::Start(keep)).await?; - Ok(Self { - out: Some(ChunkWriter::new(file.into_std().await, keep)), - records, - spooled_bytes: bytes, - ..Self::new(path, mem_max) - }) + let out = ChunkWriter::new(file.into_std().await, keep); + Ok(Self::at_file(path, mem_max, out, mark)) } /// Reopen an immutable spool at a previously acknowledged record boundary - pub async fn resume(path: PathBuf, mark: SpoolMark, offset: u64) -> Result { + pub async fn resume(path: PathBuf, mark: SpoolMark, offset: u64) -> Result { let SpoolMark { records, bytes } = mark; let length = tokio::fs::metadata(&path).await?.len(); if offset > bytes || length.checked_sub(4) != Some(bytes) { @@ -242,23 +262,53 @@ impl DeferredSpool { }); } open_validated(&path, 0).await?; - Ok(Self { + Ok(ResumedSpool { + path, records, spooled_bytes: bytes, read_offset: offset, - ..Self::new(path, 0) }) } /// Seal writes, hand back a sequential reader - pub async fn into_reader(mut self) -> Result { - if let Some(mut out) = self.out.take() { - out.flush().await?; - } else if self.spooled_bytes == 0 { - return Ok(DeferredReader { - src: ReadSrc::Mem(self.mem.into_iter()), - }); + pub async fn into_reader(self) -> Result { + let mut out = match self.store { + Store::Mem { tuples, .. } => { + return Ok(DeferredReader { + src: ReadSrc::Mem(tuples.into_iter()), + }); + } + Store::File(out) => out, + }; + out.flush().await?; + Ok(DeferredReader { + src: ReadSrc::File { + chunks: open_validated(&self.path, 0).await?, + remaining_bytes: self.spooled_bytes, + path: self.path, + }, + }) + } + + /// Drop without replay (walk failure); unlink any file + pub async fn discard(self) { + if let Store::File(_) = self.store { + let _ = tokio::fs::remove_file(&self.path).await; } + } +} + +impl ResumedSpool { + pub fn records(&self) -> u64 { + self.records + } + + pub fn spooled_bytes(&self) -> u64 { + self.spooled_bytes + } + + /// Reader past the acknowledged prefix + pub async fn into_reader(self) -> Result { Ok(DeferredReader { src: ReadSrc::File { chunks: open_validated(&self.path, self.read_offset).await?, @@ -267,13 +317,61 @@ impl DeferredSpool { }, }) } +} - /// Drop without replay (walk failure); unlink any file - pub async fn discard(mut self) { - if self.out.take().is_some() { - let _ = tokio::fs::remove_file(&self.path).await; +impl Replayable { + pub fn records(&self) -> u64 { + match self { + Self::Open(s) => s.records(), + Self::Resumed(s) => s.records(), } } + + pub fn resident_bytes(&self) -> usize { + match self { + Self::Open(s) => s.resident_bytes(), + Self::Resumed(_) => 0, + } + } + + pub fn spooled_bytes(&self) -> u64 { + match self { + Self::Open(s) => s.spooled_bytes(), + Self::Resumed(s) => s.spooled_bytes(), + } + } + + pub async fn into_reader(self) -> Result { + match self { + Self::Open(s) => s.into_reader().await, + Self::Resumed(s) => s.into_reader().await, + } + } +} + +impl From for Replayable { + fn from(spool: DeferredSpool) -> Self { + Self::Open(spool) + } +} + +impl From for Replayable { + fn from(spool: ResumedSpool) -> Self { + Self::Resumed(spool) + } +} + +/// Frame `value` onto `out`; returns bytes written +async fn append(out: &mut ChunkWriter, value: &BackfillTuple) -> Result { + let buf = &mut out.buf; + let len_at = buf.len(); + push_u32(buf, 0); + let body_at = buf.len(); + encode_record(value, buf); + let len = (buf.len() - body_at) as u32; + buf[len_at..body_at].copy_from_slice(&len.to_le_bytes()); + out.maybe_flush().await?; + Ok(4 + u64::from(len)) } /// Where the last whole record below `limit` ends, and how many there were. diff --git a/src/backfill/visibility_gate.rs b/src/backfill/visibility_gate.rs index 133e5626..54e09799 100644 --- a/src/backfill/visibility_gate.rs +++ b/src/backfill/visibility_gate.rs @@ -21,7 +21,7 @@ use crate::backfill::visibility_pending::{PendingManifest, PendingSpool}; use crate::backfill::walk_barrier::{WALK_CHECKPOINT_PERIOD, WalkBarrier}; use crate::config::ResolvedConfig; use crate::decode::visibility::{ - HEAP_XMAX_IS_MULTI, PgXactPatch, PgXactView, Visibility, deferred_xids, read_pg_multixact, + HEAP_XMAX_IS_MULTI, PgXactView, SealedPatch, Visibility, deferred_xids, read_pg_multixact, read_pg_xact, tuple_visibility, }; use crate::emit::ch_emitter::EmitterStats; @@ -329,7 +329,7 @@ impl GreenfieldSink { let row_policy = self.dest.current().row_policy(); async move { bootstrap::drain_deferred( - lane.spool, + lane.spool.into(), &catalog, &mapping, &lane.msg_tx, @@ -410,7 +410,7 @@ pub struct PendingGate { pub async fn resolve_greenfield( gate: PendingGate, data_dir: &Path, - patch: &PgXactPatch, + patch: &SealedPatch, source_major: u32, ) -> Result<(GateStats, Vec)> { let PendingGate { @@ -629,13 +629,13 @@ mod tests { ); barrier.publish_drain(2, 9, SpoolMark::default()).await; assert_eq!( - barrier.collect().await.unwrap().files, + barrier.collect().await.unwrap().files(), vec!["base/5/16400".to_string()], "the second file's tuple is still undrained" ); barrier.publish_drain(3, 9, SpoolMark::default()).await; assert_eq!( - barrier.collect().await.unwrap().files, + barrier.collect().await.unwrap().files(), vec!["base/5/16400.1".to_string()], ); } @@ -683,7 +683,7 @@ mod tests { ); barrier.publish_drain(2, 4, SpoolMark::default()).await; assert_eq!( - barrier.collect().await.unwrap().files, + barrier.collect().await.unwrap().files(), vec!["base/5/16400".to_string()] ); } @@ -992,7 +992,7 @@ mod tests { #[tokio::test] async fn several_spools_all_replay_into_one_output() { let accum = PgXactAccum::new(); - let patch = PgXactPatch::new(); + let patch = SealedPatch::default(); let multi = PgMultiXactAccum::new(17); let view = PgXactView::new(&accum, &patch).with_multixact(&multi); let (tx, mut rx) = mpsc::channel(16); @@ -1039,7 +1039,7 @@ mod tests { #[tokio::test] async fn in_flight_tuples_reach_pending() { let accum = in_progress_accum(); - let patch = PgXactPatch::new(); + let patch = SealedPatch::default(); let multi = PgMultiXactAccum::new(17); let view = PgXactView::new(&accum, &patch).with_multixact(&multi); let (tx, mut rx) = mpsc::channel(8); @@ -1079,7 +1079,7 @@ mod tests { #[tokio::test] async fn in_flight_tuples_without_pending_stay_gated() { let accum = in_progress_accum(); - let patch = PgXactPatch::new(); + let patch = SealedPatch::default(); let multi = PgMultiXactAccum::new(17); let view = PgXactView::new(&accum, &patch).with_multixact(&multi); let (tx, _rx) = mpsc::channel(8); @@ -1095,7 +1095,7 @@ mod tests { #[tokio::test] async fn undecidable_multixact_aborts_the_pass() { let accum = PgXactAccum::new(); - let patch = PgXactPatch::new(); + let patch = SealedPatch::default(); let multi = PgMultiXactAccum::new(17); let view = PgXactView::new(&accum, &patch).with_multixact(&multi); let (tx, _rx) = mpsc::channel(4); @@ -1136,10 +1136,14 @@ mod tests { ..Default::default() }, }; - let (stats, pending_tables) = - resolve_greenfield(pending, Path::new("/nonexistent"), &PgXactPatch::new(), 17) - .await - .unwrap(); + let (stats, pending_tables) = resolve_greenfield( + pending, + Path::new("/nonexistent"), + &SealedPatch::default(), + 17, + ) + .await + .unwrap(); assert_eq!(stats.emitted, 7); assert!(pending_tables.is_empty()); } diff --git a/src/backfill/visibility_pending.rs b/src/backfill/visibility_pending.rs index 58ca2590..8693a8a7 100644 --- a/src/backfill/visibility_pending.rs +++ b/src/backfill/visibility_pending.rs @@ -416,6 +416,10 @@ pub struct PendingEntry { pub committed: Vec, #[serde(default)] pub aborted: Vec, + /// Awaiting EXCHANGE: outcomes accumulate, promotion waits since the + /// swap would discard promoted rows + #[serde(default)] + pub held: bool, /// Limit promotion to newly resolved rows; relearn after crash for retry #[serde(skip)] fresh_committed: Vec, @@ -432,6 +436,10 @@ impl PendingEntry { } } + fn is(&self, rel: &RelName) -> bool { + *self.namespace == *rel.namespace && *self.relname == *rel.name + } + fn note(&mut self, xid: u32, committed: bool) -> bool { let Some(i) = self.outstanding.iter().position(|x| *x == xid) else { return false; @@ -456,6 +464,16 @@ impl PendingEntry { } } +/// Bootstrap's pending tables are durable, so its marker may clear +pub struct PendingRecorded(()); + +impl PendingRecorded { + /// No visibility gate ran, so no pending table exists + pub fn nothing_pending() -> Self { + Self(()) + } +} + /// Persist pending tables and transaction outcomes for restart recovery #[derive(Debug)] pub struct PendingLedger { @@ -548,8 +566,54 @@ impl PendingLedger { self.persist().await } + /// [`Self::push`] each manifest under one persist + pub async fn record(&mut self, manifests: &[PendingManifest]) -> io::Result { + if !manifests.is_empty() { + manifests.iter().for_each(|m| self.stage(m)); + self.persist().await?; + } + Ok(PendingRecorded(())) + } + /// [`Self::push`] without persisting pub fn stage(&mut self, manifest: &PendingManifest) { + self.stage_entry(manifest, false); + } + + /// [`Self::stage`] withholding promotion until [`Self::activate`] + pub fn hold(&mut self, manifest: &PendingManifest) { + self.stage_entry(manifest, true); + } + + /// Release a held entry once its EXCHANGE applied. Nothing promoted + /// while held, so every decided xid is fresh + pub fn activate(&mut self, rel: &RelName) -> bool { + let Some(e) = self.entries.iter_mut().find(|e| e.held && e.is(rel)) else { + return false; + }; + e.held = false; + e.fresh_committed.clone_from(&e.committed); + e.fresh_aborted.clone_from(&e.aborted); + true + } + + /// Forget a held entry whose load never published; a retry rebuilds it + pub fn discard_held(&mut self, rel: &RelName) -> bool { + let before = self.entries.len(); + self.entries.retain(|e| !e.held || !e.is(rel)); + self.entries.len() != before + } + + /// Forget held entries `swapped` rejects. Boot cleanup: a crash between + /// hold and swap mark leaves entries no swap resume will activate + pub fn discard_held_unless(&mut self, swapped: impl Fn(&RelName) -> bool) -> bool { + let before = self.entries.len(); + self.entries + .retain(|e| !e.held || swapped(&RelName::new(&e.namespace, &e.relname))); + self.entries.len() != before + } + + fn stage_entry(&mut self, manifest: &PendingManifest, held: bool) { let mut xids = manifest.xids.clone(); xids.sort_unstable(); let entry = PendingEntry { @@ -561,6 +625,7 @@ impl PendingLedger { outstanding: xids, committed: Vec::new(), aborted: Vec::new(), + held, fresh_committed: Vec::new(), fresh_aborted: Vec::new(), }; @@ -623,6 +688,9 @@ pub async fn settle( ) -> Result<(), String> { let mut done = Vec::new(); for (i, e) in ledger.entries.iter_mut().enumerate() { + if e.held { + continue; + } let rel = e.rel(); // Crash may follow DROP but precede ledger persist if sess @@ -906,6 +974,31 @@ mod tests { assert!(second.ends_with("AND (`_ws_xmax` IN (102))"), "{second}"); } + #[tokio::test] + async fn held_entry_promotes_every_outcome_once_activated() { + let tmp = tempfile::tempdir().unwrap(); + let mut ledger = PendingLedger::load(tmp.path(), 7).await.unwrap(); + ledger.hold(&manifest("orders", vec![101, 102])); + ledger.hold(&manifest("items", vec![103])); + // Outcome lands while EXCHANGE is outstanding, then restart drops fresh + ledger.note(101, &[], true); + ledger.persist().await.unwrap(); + let mut ledger = PendingLedger::load(tmp.path(), 7).await.unwrap(); + assert!(ledger.entries().iter().all(|e| e.held)); + + let orders = RelName::new("public", "orders"); + assert!(!ledger.discard_held_unless(|_| true)); + assert!(ledger.discard_held_unless(|r| *r == orders)); + assert!(!ledger.discard_held(&RelName::new("public", "items"))); + assert!(ledger.activate(&orders)); + assert!(!ledger.activate(&orders), "already active"); + assert!(!ledger.discard_held(&orders), "active entries stay"); + let e = &ledger.entries()[0]; + assert!(!e.held); + let sql = promote_sql(&e.rel(), e, "`id`"); + assert!(sql.ends_with("AND (`_ws_xmin` IN (101))"), "{sql}"); + } + #[test] fn pending_create_names_every_key_clause() { assert_eq!( @@ -946,7 +1039,7 @@ mod tests { let mut ledger = PendingLedger::load(tmp.path(), 7).await.unwrap(); ledger.push(&manifest("orders", vec![101])).await.unwrap(); let accum = PgXactAccum::new(); - let patch = PgXactPatch::new(); + let patch = PgXactPatch::new().seal(); let fold = ledger.note_view(&PgXactView::new(&accum, &patch)); assert_eq!(fold.settled, 0); assert_eq!(fold.undecidable, 1, "an aged-out xid is visible as such"); diff --git a/src/backfill/wal_landing.rs b/src/backfill/wal_landing.rs index de35f6d2..5b0f94f6 100644 --- a/src/backfill/wal_landing.rs +++ b/src/backfill/wal_landing.rs @@ -57,7 +57,7 @@ pub async fn filter_landed_wal( return Ok(LandedWalStats::default()); }; - let mut stream = WalStream::new( + let mut stream = WalStream::builder( timeline, WAL_SEG_SIZE, Pos::new(first.start_lsn(WAL_SEG_SIZE)), @@ -75,6 +75,7 @@ pub async fn filter_landed_wal( ) .await?; } + let mut stream = stream.start(); let mut records = DropRecords; let mut writer = WriteBack { diff --git a/src/backfill/wal_replay.rs b/src/backfill/wal_replay.rs index 335a6512..9bc5a452 100644 --- a/src/backfill/wal_replay.rs +++ b/src/backfill/wal_replay.rs @@ -22,7 +22,7 @@ use crate::decode::wal_xact::{ XLOG_XACT_COMMIT_PREPARED, XLOG_XACT_OPMASK, parse_xact_assignment, parse_xact_payload, }; use crate::emit::ch_emitter::EmitterStats; -use crate::emit::pipeline::ack::AckHandle; +use crate::emit::pipeline::ack::{AckHandle, OpenSeq, Publish, SeqAlloc}; use crate::emit::pipeline::batcher::{BatcherMsg, RoutedRow}; use crate::emit::route::{RouteSnapshot, freeze_routes}; use crate::mapping::MappingSnapshot; @@ -101,10 +101,10 @@ pub struct WalReplaySink { batch_rows: usize, batch_bytes: usize, msg_tx: mpsc::Sender, - ack: AckHandle, patch: Option>>, - /// Current `(sequence, routed rows)`, registered on first row - open: Option<(u64, u64)>, + seqs: SeqAlloc, + /// Current seq and its routed rows, registered on first row + open: Option<(OpenSeq, u64)>, /// Rows waiting for next mirror write pending_rows: Vec, pending_bytes: usize, @@ -146,17 +146,14 @@ impl WalReplaySink { batch_rows: inputs.batch_rows, batch_bytes: inputs.batch_bytes, msg_tx: inputs.msg_tx, - ack: inputs.ack, patch: inputs.patch, + seqs: inputs.ack.seqs(inputs.next_seq), open: None, pending_rows: Vec::new(), pending_bytes: 0, pending_permits: Vec::new(), pending_cap, - replay: ReplayStats { - next_seq: inputs.next_seq, - ..Default::default() - }, + replay: ReplayStats::default(), } } @@ -173,14 +170,17 @@ impl WalReplaySink { } pub fn stats(&self) -> ReplayStats { - self.replay + ReplayStats { + next_seq: self.seqs.next(), + ..self.replay + } } /// Mirror writes flushed, seq boundary reported. Commits close their own /// seq, so a segment boundary leaves none open pub async fn segment_boundary(&mut self) -> std::result::Result { self.flush_rows().await?; - Ok(self.replay.next_seq) + Ok(self.seqs.next()) } /// Lowest first-record LSN still buffered. A resume above it would drop @@ -333,7 +333,7 @@ impl WalReplaySink { .commit(xid, &payload.subxacts); } // Resolve filenodes invisible at record time before drain - resolve_stash( + let stash = resolve_stash( &self.buffer, &self.log, &self.pending, @@ -349,7 +349,7 @@ impl WalReplaySink { .lock() .await .drain_committed( - xid, + stash, payload.xact_time, record.source_lsn, &payload.subxacts, @@ -367,7 +367,7 @@ impl WalReplaySink { } drain.finish().await.map_err(SinkError::from)?; if let Some((seq, rows)) = self.open.take() { - self.ack.placed(seq, rows); + seq.place(rows); } self.subxact_tracker.forget_tree(xid); Ok(()) @@ -397,7 +397,7 @@ impl WalReplaySink { // no block ref, so never passes the rfn filter WalkStep::Event(DrainEntry::Catalog(_)) | WalkStep::Event(DrainEntry::Config(_)) - | WalkStep::Truncate(_) => {} + | WalkStep::Truncate { .. } => {} WalkStep::Event(DrainEntry::ToastBarrier { toast_relid, marker_lsn, @@ -453,16 +453,11 @@ impl WalReplaySink { let value_permit = detoast_heap(&mut heap, spool, &ref_maps, &self.resolver) .await .map_err(SinkError::from)?; - let seq = if let Some((seq, rows)) = &mut self.open { - *rows += 1; - *seq - } else { - let seq = self.replay.next_seq; - self.replay.next_seq += 1; - self.ack.register(seq, commit_lsn); - self.open = Some((seq, 1)); - seq - }; + let (open, rows) = self + .open + .get_or_insert_with(|| (self.seqs.open(commit_lsn, Publish::Commit), 0)); + *rows += 1; + let seq = open.seq(); self.msg_tx .send(BatcherMsg::Row(RoutedRow { seq, @@ -578,7 +573,7 @@ pub struct SegmentPump { impl SegmentPump { pub fn start(first: &SegmentName, target_db_oid: Oid) -> Result { - let mut stream = WalStream::new( + let mut stream = WalStream::builder( first.timeline, WAL_SEG_SIZE, Pos::new(first.start_lsn(WAL_SEG_SIZE)), @@ -586,7 +581,7 @@ impl SegmentPump { .map_err(|e| anyhow::anyhow!("wal_replay: WalStream: {e}"))?; stream.filter_mut().set_target_db(target_db_oid); Ok(Self { - stream, + stream: stream.start(), seg_sink: DropSegments, }) } @@ -638,7 +633,7 @@ pub async fn pump_segments_through( mod tests { use super::*; use crate::catalog::desc_log::DescLogIdentity; - use crate::decode::visibility::{PgXactAccum, PgXactView, XidStatus}; + use crate::decode::visibility::{PgXactAccum, PgXactView, SealedPatch, XidStatus}; use crate::decode::wal_xact::{ XACT_XINFO_HAS_SUBXACTS, XACT_XINFO_HAS_TWOPHASE, XLOG_XACT_HAS_INFO, }; @@ -962,7 +957,7 @@ mod tests { }) } - fn status(patch: &PgXactPatch, xid: u32) -> XidStatus { + fn status(patch: &SealedPatch, xid: u32) -> XidStatus { let accum = PgXactAccum::new(); PgXactView::new(&accum, patch).xid_status(xid) } @@ -981,7 +976,7 @@ mod tests { )) .await .unwrap(); - let patch = patch.lock().unwrap(); + let patch = std::mem::take(&mut *patch.lock().unwrap()).seal(); assert_eq!(status(&patch, PREPARED_XID), XidStatus::Committed); assert_eq!(status(&patch, PREPARED_SUBXID), XidStatus::Committed); assert_ne!(status(&patch, FINISHER_XID), XidStatus::Committed); @@ -1000,7 +995,7 @@ mod tests { )) .await .unwrap(); - let patch = patch.lock().unwrap(); + let patch = std::mem::take(&mut *patch.lock().unwrap()).seal(); assert_eq!(status(&patch, PREPARED_XID), XidStatus::Aborted); assert_ne!(status(&patch, FINISHER_XID), XidStatus::Aborted); } diff --git a/src/backfill/walk_barrier.rs b/src/backfill/walk_barrier.rs index 8951e73d..46d8ae12 100644 --- a/src/backfill/walk_barrier.rs +++ b/src/backfill/walk_barrier.rs @@ -22,6 +22,7 @@ use std::time::Duration; use tokio::sync::Mutex; use crate::backfill::spool::SpoolMark; +use crate::emit::pipeline::tail::OwnedTail; /// Seconds between stage publishes and checkpoint writes pub const WALK_CHECKPOINT_PERIOD: Duration = Duration::from_secs(30); @@ -47,10 +48,38 @@ struct PartTally { /// Marks and file names a checkpoint may record once the tail proves /// `next_seq` pub struct WalkProof { - pub files: Vec, - pub gate_deferred: SpoolMark, - pub toast_deferred: SpoolMark, - pub next_seq: u64, + files: Vec, + gate_deferred: SpoolMark, + toast_deferred: SpoolMark, + next_seq: u64, +} + +/// [`WalkProof`] whose seqs the tail proved durable +pub struct DurableWalk(WalkProof); + +impl WalkProof { + #[cfg(test)] + pub(crate) fn files(&self) -> &[String] { + &self.files + } + + pub async fn prove(self, tail: &OwnedTail) -> Result { + tail.checkpoint(self.next_seq).await?; + Ok(DurableWalk(self)) + } +} + +impl DurableWalk { + /// Files walked, then gate and drain spool marks + pub fn into_parts(self) -> (Vec, SpoolMark, SpoolMark) { + let WalkProof { + files, + gate_deferred, + toast_deferred, + .. + } = self.0; + (files, gate_deferred, toast_deferred) + } } #[derive(Default)] diff --git a/src/bin/stream/bootstrap.rs b/src/bin/stream/bootstrap.rs index 79f5f06e..32cf3d6e 100644 --- a/src/bin/stream/bootstrap.rs +++ b/src/bin/stream/bootstrap.rs @@ -38,6 +38,7 @@ use walshadow::schema::{RelName, SchemaEvent}; use walshadow::source_feed::SourceFeed; use walshadow::toast::ToastResolver; use walshadow::visibility::PgXactPatch; +use walshadow::visibility_pending::{PendingLedger, PendingRecorded}; use crate::archive::fetch_wal_into_pg_wal; use crate::args::{Args, cli_base}; @@ -162,10 +163,6 @@ pub(crate) async fn run_bootstrap( .await .context("bootstrap: seed catalog filenodes")?; let catalog_filenodes: Vec<_> = landing_tracker.nodes().collect(); - // Filtered at one of two points depending on toast mode, never both: - // shadow mode rewrites before its recovery starts mid-bootstrap, other - // modes after the window leg has read the raw segments - let mut landing_tracker = Some(landing_tracker); tracing::info!( target: "walshadow::bootstrap", relations = catalog_map.len(), @@ -473,7 +470,11 @@ pub(crate) async fn run_bootstrap( let window_scratch = args.spill_dir.join("bootstrap_window"); tokio::fs::remove_dir_all(&window_scratch).await.ok(); - let (shipped, outcome, window) = if let Some(target) = ch_target { + // Landed WAL is filtered at one of two points, never both: shadow mode + // rewrites before its recovery starts mid-bootstrap, other modes after the + // window leg has read the raw segments. The branch hands back the tracker + // it left unused + let (shipped, outcome, window, landing_tracker) = if let Some(target) = ch_target { let (emitter_cfg, mapping, resolved, skip_initial) = target; // Route bootstrap rows through the shared insert tail. Bootstrap // is the easy case: every row op=Insert at _lsn = start_lsn, no @@ -809,12 +810,14 @@ pub(crate) async fn run_bootstrap( // Rewrite landed WAL before shadow recovery; backup processing uses copy. // // Non-shadow toast modes rewrite after reading original `pg_wal` below - if shadow_toast { + let landing_tracker = if !shadow_toast { + Some(landing_tracker) + } else { let landed = walshadow::backfill::wal_landing::filter_landed_wal( &shadow_data_dir.join("pg_wal"), outcome.start.timeline, outcome.end.end_lsn, - landing_tracker.take().expect("landed WAL filtered once"), + landing_tracker, Some((shadow_toast_rels, outcome.start.start_lsn)), ) .await @@ -877,7 +880,8 @@ pub(crate) async fn run_bootstrap( .set(Arc::new(bridge)) .ok() .context("bootstrap: shadow TOAST bridge bound twice")?; - } + None + }; // Replay backup WAL before resolving deferred value references if let Some(mut cfg) = window_cfg { @@ -977,7 +981,7 @@ pub(crate) async fn run_bootstrap( oracle: oracle.clone(), stream_stats: gate_stats, }); - (rows_routed, outcome, window) + (rows_routed, outcome, window, landing_tracker) } else { // Metrics-only skips destination convergence let mut observer = MetricsTupleObserver::default(); @@ -986,7 +990,7 @@ pub(crate) async fn run_bootstrap( let outcome: BootstrapOutcome = pump_res .context("bootstrap pump join")? .context("bootstrap pump")?; - (shipped, outcome, None) + (shipped, outcome, None, Some(landing_tracker)) }; // Replace live-leg COPY connection and recheck source identity @@ -1071,7 +1075,7 @@ pub(crate) async fn run_bootstrap( tokio::fs::remove_dir_all(&window_scratch).await.ok(); // Non-shadow toast modes rewrite landed WAL after backup processing reads it - if let Some(tracker) = landing_tracker.take() { + if let Some(tracker) = landing_tracker { let landed = walshadow::backfill::wal_landing::filter_landed_wal( &shadow_data_dir.join("pg_wal"), outcome.start.timeline, @@ -1093,16 +1097,15 @@ pub(crate) async fn run_bootstrap( } // Resolve deferred tuples after window transaction overlay is complete - if let Some(pending) = pending_gate { - let mut patch = std::mem::take(&mut *window_patch.lock().expect("window patch lock")); - patch.seal(); + let recorded = if let Some(pending) = pending_gate { + let patch = std::mem::take(&mut *window_patch.lock().expect("window patch lock")).seal(); let (gate, pending_tables) = resolve_greenfield(pending, &shadow_data_dir, &patch, source_major) .await .context("bootstrap: visibility gate")?; // Persist the ledger before clearing the marker so pending rows // already in ClickHouse can be published after restart - let mut ledger = walshadow::visibility_pending::PendingLedger::load( + let mut ledger = PendingLedger::load( &args.spill_dir, source_ident .sysid @@ -1111,12 +1114,10 @@ pub(crate) async fn run_bootstrap( ) .await .context("bootstrap: load pending visibility ledger")?; - for m in &pending_tables { - ledger - .push(m) - .await - .context("bootstrap: persist pending visibility ledger")?; - } + let recorded = ledger + .record(&pending_tables) + .await + .context("bootstrap: persist pending visibility ledger")?; tracing::info!( target: "walshadow::bootstrap", emitted = gate.emitted, @@ -1129,7 +1130,10 @@ pub(crate) async fn run_bootstrap( patch_xacts = patch.len(), "bootstrap visibility gate settled", ); - } + recorded + } else { + PendingRecorded::nothing_pending() + }; // PG refuses to start on a data dir whose mode isn't 0700 or 0750. // BASE_BACKUP tar carries no entry for the root, so extraction leaves @@ -1143,8 +1147,7 @@ pub(crate) async fn run_bootstrap( .with_context(|| format!("bootstrap: chmod 0700 {}", shadow_data_dir.display()))?; } - bootstrap_marker::ExtractedCheckpoint::clear(&shadow_data_dir).await?; - BootstrapMarker::clear(&shadow_data_dir).await?; + BootstrapMarker::complete(&shadow_data_dir, recorded).await?; timing.finish(); Ok(( @@ -1222,6 +1225,7 @@ pub(crate) async fn bootstrap_build_mapping( args.ch_config.clone(), cli_base(args), mapping.clone(), + walshadow::config::ResolverBoot::default(), ); let (ddl_cfg, merged_tables, resolved) = { let snap = config_rx.borrow(); diff --git a/src/bin/stream/housekeeping.rs b/src/bin/stream/housekeeping.rs index 5db3ffb0..10cc263b 100644 --- a/src/bin/stream/housekeeping.rs +++ b/src/bin/stream/housekeeping.rs @@ -40,15 +40,16 @@ pub(crate) fn spawn_segment_fsync( return; } }; - while let Some(item) = rx.recv().await { - let mut max_lsn = item.end_lsn; + while let Some(mut last) = rx.recv().await { while let Ok(next) = rx.try_recv() { - max_lsn = max_lsn.max(next.end_lsn); + if next.end_lsn() > last.end_lsn() { + last = next; + } } let dir = Arc::clone(&dir); - let synced = tokio::task::spawn_blocking(move || walshadow::fs::syncfs(&dir)).await; + let synced = tokio::task::spawn_blocking(move || last.syncfs(&dir)).await; match synced { - Ok(Ok(())) => {} + Ok(Ok(durable)) => durable_lsn.publish(durable), Ok(Err(e)) => { fatal.set(format!("syncfs {}: {e}", out_dir.display())); return; @@ -57,8 +58,7 @@ pub(crate) fn spawn_segment_fsync( fatal.set(format!("syncfs join {}: {e}", out_dir.display())); return; } - } - durable_lsn.join(Pos::new(max_lsn)); + }; } }) } @@ -107,20 +107,24 @@ pub(crate) async fn trim_retention( if lsn.is_zero() { continue; } - if client.is_none() { - match open_retention_client(&shadow_conninfo).await { - Ok(c) => client = Some(c), - Err(e) => { + let conn = if let Some(c) = &mut client { + c + } else { + let Ok(c) = open_retention_client(&shadow_conninfo) + .await + .inspect_err(|e| { tracing::warn!( target: "walshadow::retention", error = %e, "shadow connect failed; retrying next cycle", ); - continue; - } - } - } - let redo = match query_redo_lsn(client.as_ref().expect("just set")).await { + }) + else { + continue; + }; + client.insert(c) + }; + let redo = match query_redo_lsn(conn).await { Ok(v) => v, Err(e) => { tracing::warn!(target: "walshadow::retention", error = %e, "redo lsn query"); diff --git a/src/bin/stream/metrics_publish.rs b/src/bin/stream/metrics_publish.rs index abff654a..29b26689 100644 --- a/src/bin/stream/metrics_publish.rs +++ b/src/bin/stream/metrics_publish.rs @@ -284,7 +284,7 @@ pub(crate) async fn populate_metrics( pause_consumed_lsn: timeline_view.pause_frontier.map_or(0, |(c, _)| c), pause_received_lsn: timeline_view.pause_frontier.map_or(0, |(_, r)| r), pause_refrozen: timeline_view.pause_refrozen, - promotion_ready: timeline_view.promotion.ready, + promotion_ready: timeline_view.promotion.ready(), promotion_blocked_on: timeline_view.promotion.blocked_on, promotion_target_in_recovery: timeline_view.promotion.in_recovery, promotion_target_replay_lsn: timeline_view.promotion.replay_lsn, diff --git a/src/bin/stream/runtime_cfg.rs b/src/bin/stream/runtime_cfg.rs index a50fe722..8f1d5faa 100644 --- a/src/bin/stream/runtime_cfg.rs +++ b/src/bin/stream/runtime_cfg.rs @@ -7,7 +7,7 @@ use ahash::HashSet; use anyhow::Context; use tokio::sync::{Mutex, watch}; use tokio_util::sync::CancellationToken; -use walshadow::config::{ConfigResolver, ResolvedConfig}; +use walshadow::config::ResolvedConfig; use walshadow::mapping::MappingHandle; use walshadow::pg::quote_ident; use walshadow::runtime_config::InitialLoadMode; @@ -72,15 +72,17 @@ pub(crate) async fn or_signal( } } -/// Seed the resolver overlay from source PG's `.config_*` tables via -/// the sidecar libpq connection (plan §7). Refuses (Err → daemon exits) when -/// the schema is named but not installed, or the install is newer than this -/// daemon understands — explicit opt-in should not silently no-op. -pub(crate) async fn seed_runtime_config( +/// Load the resolver's boot overlay from source PG's `.config_*` +/// tables via the sidecar libpq connection (plan §7). Refuses (Err → daemon +/// exits) when the schema is named but not installed, or the install is newer +/// than this daemon understands — explicit opt-in should not silently no-op. +pub(crate) async fn load_runtime_config( client: &tokio_postgres::Client, schema: &str, - resolver: &ConfigResolver, -) -> anyhow::Result> { +) -> anyhow::Result<( + walshadow::runtime_config::ConfigOverlay, + Vec<(RelName, walshadow::runtime_config::TableRow)>, +)> { use walshadow::runtime_config::{ColumnRow, ConfigOverlay, GlobalRow, NamespaceRow, TableRow}; let s = quote_ident(schema); let mut overlay = ConfigOverlay::default(); @@ -203,7 +205,6 @@ pub(crate) async fn seed_runtime_config( .iter() .map(|(k, v)| (k.clone(), v.clone())) .collect(); - resolver.seed_overlay(overlay).await; tracing::info!( target: "walshadow::config", schema, @@ -213,7 +214,7 @@ pub(crate) async fn seed_runtime_config( columns = n_col, "runtime config overlay seeded from source PG", ); - Ok(table_rows) + Ok((overlay, table_rows)) } pub(crate) async fn apply_toml_initial_loads( diff --git a/src/bin/stream/session.rs b/src/bin/stream/session.rs index 76da235e..5930ab43 100644 --- a/src/bin/stream/session.rs +++ b/src/bin/stream/session.rs @@ -20,7 +20,8 @@ use walshadow::metrics::{MetricsRegistry, RateEstimator}; use walshadow::pg::socket_conninfo; use walshadow::pipeline::{PipelineConfig, TailKind}; use walshadow::pos::{ - EmitterAck, FilterDurable, Floor, Monotone, Pos, ShadowFlush, ShadowReplay, SourceReceived, + EmitterAck, FilterDurable, Floor, Monotone, Pos, RawStart, ShadowFlush, ShadowReplay, + SourceReceived, }; use walshadow::queueing_record_sink::{ DEFAULT_QUEUEING_BATCH_SIZE, DEFAULT_QUEUEING_RECORD_SINK_CAPACITY, QueueingRecordSink, @@ -61,9 +62,10 @@ use crate::source_db::{ open_source_sql_client, }; use crate::source_recovery::{ - BARRIER_LOG_INTERVAL, FORK_FENCE_DRAIN, PROMOTION_POLL, PromotionGate, ReconnectBackoff, - SOURCE_SWAP_RETRY, SourcePath, SourceRecovery, commit_fork_resume, connect_source_waiting, - promotion_gate, resume_manifest, resume_source_feed, stream_branch, swap_reason, + BARRIER_LOG_INTERVAL, FORK_FENCE_DRAIN, PROMOTION_POLL, PauseState, PromotionGate, + ReconnectBackoff, SOURCE_SWAP_RETRY, SourcePath, SourceRecovery, commit_fork_resume, + connect_source_waiting, promotion_gate, resume_manifest, resume_source_feed, stream_branch, + swap_reason, }; pub(crate) async fn run_session( @@ -274,7 +276,7 @@ pub(crate) async fn run_session( .and_then(|c| c.backup.clone()) .map(Archive::open) .transpose()?; - let start_lsn_override: Option> = args + let start_lsn_override: Option> = args .start_lsn .as_deref() .map(|s| walshadow::pg::parse_pg_lsn(s).context("--start-lsn")) @@ -462,7 +464,7 @@ pub(crate) async fn run_session( .filter(|tli| *tli <= start_timeline) .collect(); - let mut stream = WalStream::new(start_timeline, WAL_SEG_SIZE, aligned)?; + let mut stream = WalStream::builder(start_timeline, WAL_SEG_SIZE, aligned)?; let prefix_dirs = [args.out_dir.clone(), shadow_start.data_dir().join("pg_wal")]; stream.preserve_resume_prefix(&prefix_dirs).await?; // Shadow must attach to this listener before catalog replay can advance @@ -733,7 +735,8 @@ pub(crate) async fn run_session( .map(|c| c.pending_capture) .unwrap_or_default(); let pending_catalog = Arc::new(walshadow::pending::PendingCatalog::default()); - let smgr_markers = stream.filter_mut().smgr_markers(); + let smgr_markers = stream.filter().smgr_markers(); + let mut stream = stream.start(); // One log per database, each in its own spill subdirectory; the primary // keeps the spill root so a single-database resume reads where it wrote let mut desc_logs: Vec> = @@ -988,23 +991,18 @@ pub(crate) async fn run_session( } }; - let (mut reorder_sink, pipeline_handle) = pcfg + let (booting_reorder, pipeline_handle) = pcfg .spawn(emitter_ack.clone()) .await .context("spawn decode+insert pipeline")?; let ack_probe = pipeline_handle.ack_probe.clone(); - reorder_sink - .flush_due_retires() - .await - .context("boot flush of due toast-mirror retires")?; - reorder_sink - .settle_pending_boot(Some(&args.bootstrap_shadow_data_dir)) - .await - .context("boot settle of pending backup rows")?; - reorder_sink - .apply_boot_events(desc_log.active_present_at(raw_start.get()), raw_start.get()) - .await - .context("boot Added pass over descriptor log")?; + let reorder_sink = booting_reorder + .boot( + Some(&args.bootstrap_shadow_data_dir), + desc_log.active_present_at(raw_start.get()), + raw_start.get(), + ) + .await?; let decoder_xact = QueueingRecordSink::spawn( DecoderXactPair { decoder, @@ -1211,17 +1209,13 @@ pub(crate) async fn run_session( // resolver is the one the pump watches let pump_config_rx = config_resolvers.first().map(|r| r.subscribe()); let mut swap = SourceSwap::default(); - // Frozen when the pump observes a pause, so a promotion decision reads a - // frontier that cannot move under it. Cleared on resume: a value left over + // Frontier frozen when the pump observes a pause, so a promotion decision + // reads one that cannot move under it. Cleared on resume: a value left over // from an earlier pause is as misleading as a live one - let mut pause_frontier: Option<(u64, u64)> = None; - // A restart mid-pause re-freezes both numbers, conservatively but not - // identically, so the pair an operator already read has to be read again - let mut pause_refrozen = false; + let mut pause = PauseState::Running; let mut ever_unpaused = false; // Step 5's answer, refreshed while paused off the endpoint the pump holds - let mut promotion = PromotionGate::default(); - let mut promotion_polled_at: Option = None; + let mut promotion = PromotionGate::blocked("not_paused"); let switchover = Switchover { system_id: live_identity.system_id, out_dir: &args.out_dir, @@ -1331,45 +1325,47 @@ pub(crate) async fn run_session( // about, which the target must reach before promotion. Bytes cannot // have been consumed without being received, so a source that has not // reported a head yet reads as level with the consumed frontier - match (paused, pause_frontier) { - (true, None) => { - pause_frontier = Some(( - stream.next_lsn().get(), - received.get().max(stream.next_lsn().get()), - )); - // A pause this process never saw lifted was taken before it - // booted, so these two numbers replace ones an operator may - // already hold. Both re-freeze conservatively — consumed drops - // back to the floor, received re-derives from the live head — - // but a promotion decision has to be taken from the pair on - // offer now (architecture/recovery.md) - pause_refrozen = !ever_unpaused; - let (consumed, head) = pause_frontier.expect("just frozen"); - tracing::info!( - target: "walshadow", - pause_consumed_lsn = %format_pg_lsn(consumed), - pause_received_lsn = %format_pg_lsn(head), - refrozen = pause_refrozen, - "pause observed — frontier frozen", - ); - } - (false, Some(_)) => { - pause_frontier = None; - pause_refrozen = false; - } - _ => {} + if !paused { + pause = PauseState::Running; + } else if let PauseState::Running = pause { + let consumed = stream.next_lsn().get(); + let head = received.get().max(consumed); + // A pause this process never saw lifted was taken before it + // booted, so these two numbers replace ones an operator may + // already hold. Both re-freeze conservatively — consumed drops + // back to the floor, received re-derives from the live head — + // but a promotion decision has to be taken from the pair on + // offer now (architecture/recovery.md) + let refrozen = !ever_unpaused; + tracing::info!( + target: "walshadow", + pause_consumed_lsn = %format_pg_lsn(consumed), + pause_received_lsn = %format_pg_lsn(head), + refrozen, + "pause observed — frontier frozen", + ); + pause = PauseState::Paused { + frontier: (consumed, head), + refrozen, + polled_at: None, + }; } ever_unpaused |= !paused; // Step 5 of the protocol, answered off the connection step 4's repoint // already moved onto the target: replay, receive, and recovery state // beside the frozen frontier they have to reach // (architecture/recovery.md) - if let Some((_, pause_received)) = pause_frontier { - if promotion_polled_at.is_none_or(|t| t.elapsed() >= PROMOTION_POLL) { - promotion_polled_at = Some(Instant::now()); + if let PauseState::Paused { + frontier: (_, pause_received), + polled_at, + .. + } = &mut pause + { + if polled_at.is_none_or(|t| t.elapsed() >= PROMOTION_POLL) { + *polled_at = Some(Instant::now()); promotion = match tokio::time::timeout( PROMOTION_POLL, - promotion_gate(&mut feed, pause_received), + promotion_gate(&mut feed, *pause_received), ) .await { @@ -1382,7 +1378,6 @@ pub(crate) async fn run_session( } } else { promotion = PromotionGate::blocked("not_paused"); - promotion_polled_at = None; } let (shadow_agg, shadow_served_tli) = { let state = shadow_state.lock().await; @@ -1421,17 +1416,15 @@ pub(crate) async fn run_session( }, ); if last_cursor_write.is_none_or(|t| t.elapsed() >= cursor_write_interval) { - manifest::write(&args.spill_dir, &cur) + let persisted = manifest::write(&args.spill_dir, &cur) .await .context("write resume manifest")?; last_cursor_write = Some(Instant::now()); - // Publish only after persist: pruners cut against what a - // crash-now restart actually resumes from. - resume_floor.join(cur.floor); + resume_floor.publish(persisted); // Descriptor log prunes against the same floor, off this task: a // compaction rewrites the whole ckpt inline and would stall WAL // consumption past the source's wal_sender_timeout - gc_floor.join(cur.floor); + gc_floor.publish(persisted); } // flush caps physical slot's restart_lsn. // Manifest writes are cadence-gated above while keepalive replies inside @@ -1706,27 +1699,32 @@ pub(crate) async fn run_session( // so a restart from it loses nothing. The loop keeps publishing // meanwhile, so a wait reads as a wait rather than a stall, and the // source has stopped producing so nothing queues up behind it - let waiting_on = walshadow::transition::ForkBarrier { + let open = match (walshadow::transition::ForkBarrier { resume_safe_lsn: resume_safe, shadow_apply_lsn: shadow_agg.min_apply_lsn, filter_durable: durable, floor: resume_floor.get(), - } - .pending(Pos::new(probed.switch_lsn), WAL_SEG_SIZE); - if let Some(wait) = waiting_on { - // Prod the walreceiver: non-forced replies fire only on flush - // progress, and the ancestor's tail may be the last thing left - shadow_state.lock().await.request_status(); - if barrier_logged.is_none_or(|t| t.elapsed() >= BARRIER_LOG_INTERVAL) { - tracing::info!( - target: "walshadow", - switch_lsn = %format_pg_lsn(probed.switch_lsn), - waiting_on = wait.label(), - "fork barrier: {wait}", - ); - barrier_logged = Some(Instant::now()); + }) + .open(Pos::new(probed.switch_lsn), WAL_SEG_SIZE) + { + Ok(open) => Some(open), + Err(wait) => { + // Prod the walreceiver: non-forced replies fire only on flush + // progress, and the ancestor's tail may be the last thing left + shadow_state.lock().await.request_status(); + if barrier_logged.is_none_or(|t| t.elapsed() >= BARRIER_LOG_INTERVAL) { + tracing::info!( + target: "walshadow", + switch_lsn = %format_pg_lsn(probed.switch_lsn), + waiting_on = wait.label(), + "fork barrier: {wait}", + ); + barrier_logged = Some(Instant::now()); + } + None } - } else { + }; + if let Some(open) = open { barrier_logged = None; let commit = async |resume: walshadow::transition::ForkResume| { commit_fork_resume( @@ -1735,7 +1733,7 @@ pub(crate) async fn run_session( resume, manifest::LsnSet { // Fork cannot precede last observed source head - source_received: received.max(resume.switch_lsn.retag()), + source_received: received.max(resume.switch_lsn().retag()), filter_durable: durable, shadow_replay, drain: guards.drain_lsn, @@ -1757,6 +1755,7 @@ pub(crate) async fn run_session( status, guards, &probed, + open, commit, &mut timeline_stats, ) @@ -1882,8 +1881,8 @@ pub(crate) async fn run_session( shadow_replay_timeline: shadow_agg.replay_timeline.unwrap_or(0), floor_lsn: published_floor, stats: timeline_stats, - pause_frontier, - pause_refrozen, + pause_frontier: pause.frontier(), + pause_refrozen: pause.refrozen(), wedge: crossing.wedge().cloned(), promotion, }, diff --git a/src/bin/stream/source_db.rs b/src/bin/stream/source_db.rs index 3d168acb..8c02700e 100644 --- a/src/bin/stream/source_db.rs +++ b/src/bin/stream/source_db.rs @@ -9,15 +9,15 @@ use anyhow::Context; use tokio::sync::Mutex; use walrus::pg::backup::format_pg_lsn; use walshadow::ch_emitter::{EmitterConfig, EmitterStats}; -use walshadow::config::{CliOverrides, ConfigResolver, SourceConn}; -use walshadow::pos::{Floor, Pos}; +use walshadow::config::{CliOverrides, ConfigResolver, ResolverBoot, SourceConn}; +use walshadow::pos::{Floor, Pos, RawStart}; use walshadow::runtime_config::InitialLoadMode; use walshadow::schema::{RelName, SchemaEvent}; use walshadow::shadow_catalog::ShadowCatalog; use walshadow::source_db::{DbLink, SourceDb}; use crate::args::{Args, cli_base}; -use crate::runtime_cfg::{apply_toml_initial_loads, refresh_mapping, seed_runtime_config}; +use crate::runtime_cfg::{apply_toml_initial_loads, load_runtime_config, refresh_mapping}; use crate::session::SessionTasks; pub(crate) struct SourceDbInputs<'a> { @@ -37,7 +37,7 @@ pub(crate) struct SourceDbInputs<'a> { pub(crate) budget: &'a walshadow::budget::MemoryBudget, pub(crate) stats: &'a Arc, pub(crate) source_major: u32, - pub(crate) raw_start: Pos, + pub(crate) raw_start: Pos, /// Relations shadow can serve TOAST for, absent unless `[toast] mode = shadow` pub(crate) shadow_toast_held: Option<&'a walshadow::filter::shadow_relations::ShadowHeld>, pub(crate) system_id: u64, @@ -78,15 +78,33 @@ pub(crate) async fn build_source_db(input: SourceDbInputs<'_>) -> anyhow::Result .map(std::time::Duration::from_millis), source_slot: args.slot.clone(), }; + // Runtime-config overlay (§7): seed the resolver from this database's + // config_* tables over a sidecar libpq connection. Post-seed writes + // arrive live off the WAL stream. Refuse to start if the named schema is + // not installed — explicit opt-in means the operator expects the overlay + // present. + let mut overlay = walshadow::runtime_config::ConfigOverlay::default(); + let mut seeded_table_rows: Vec<(RelName, walshadow::runtime_config::TableRow)> = Vec::new(); + if let Some(schema) = cfg.runtime_config_schema.clone() { + let client = open_source_sql_client(input.source, &conn.name) + .await + .context("sidecar sql for runtime-config seed")?; + (overlay, seeded_table_rows) = load_runtime_config(&client, &schema) + .await + .context("seed runtime config overlay")?; + } let (resolver, config_rx) = ConfigResolver::new( &cfg, cli_overrides, args.ch_config.clone(), cli_base(args), mapping.clone(), + ResolverBoot { + overlay, + shadow_toast: input.shadow_toast_held.cloned(), + }, ); if let Some(held) = input.shadow_toast_held { - resolver.bind_shadow_toast(held.clone()); // Check configured tables here because they bypass opt-in // Preserve exclusions across SIGHUP reloads let descs = conn @@ -105,20 +123,6 @@ pub(crate) async fn build_source_db(input: SourceDbInputs<'_>) -> anyhow::Result "mapping refresher", refresh_mapping(config_rx.clone(), mapping.clone()), ); - // Runtime-config overlay (§7): before the pump consumes WAL, seed the - // resolver from this database's config_* tables over a sidecar libpq - // connection. Post-seed writes arrive live off the WAL stream. Refuse - // to start if the named schema is not installed — explicit opt-in - // means the operator expects the overlay present. - let mut seeded_table_rows: Vec<(RelName, walshadow::runtime_config::TableRow)> = Vec::new(); - if let Some(schema) = cfg.runtime_config_schema.clone() { - let client = open_source_sql_client(input.source, &conn.name) - .await - .context("sidecar sql for runtime-config seed")?; - seeded_table_rows = seed_runtime_config(&client, &schema, &resolver) - .await - .context("seed runtime config overlay")?; - } { let rc = config_rx.borrow(); cfg.row_budget = rc.row_budget; @@ -318,8 +322,8 @@ pub(crate) struct DescLogInputs<'a> { pub(crate) identity: walshadow::desc_log::DescLogIdentity, pub(crate) lineage: &'a [u32], pub(crate) manifest_present: bool, - pub(crate) start_lsn_override: Option>, - pub(crate) raw_start: Pos, + pub(crate) start_lsn_override: Option>, + pub(crate) raw_start: Pos, pub(crate) aligned: Pos, } @@ -355,7 +359,7 @@ pub(crate) async fn open_db_desc_log( ); if let Some(lsn) = input.start_lsn_override { anyhow::ensure!( - lsn >= desc_log.floor_at_write(), + lsn.get() >= desc_log.floor_at_write().get(), "--start-lsn {} below descriptor log floor {}; no shape history \ survives there — --ignore-cursor or re-bootstrap", lsn, diff --git a/src/bin/stream/source_recovery.rs b/src/bin/stream/source_recovery.rs index 8980211f..c0550a4c 100644 --- a/src/bin/stream/source_recovery.rs +++ b/src/bin/stream/source_recovery.rs @@ -42,9 +42,8 @@ pub(crate) fn stream_branch( /// Step 5's gate: what the promotion target owes before it may be promoted, /// answered off the source connection walshadow already holds rather than a /// second `psql` (architecture/recovery.md). -#[derive(Debug, Clone, Copy, Default)] +#[derive(Debug, Clone, Copy)] pub(crate) struct PromotionGate { - pub(crate) ready: bool, /// Term that fails, empty once ready pub(crate) blocked_on: &'static str, pub(crate) in_recovery: bool, @@ -56,15 +55,47 @@ impl PromotionGate { pub(crate) fn blocked(blocked_on: &'static str) -> Self { Self { blocked_on, - ..Self::default() + in_recovery: false, + replay_lsn: 0, + receive_lsn: 0, } } + pub(crate) fn ready(&self) -> bool { + self.blocked_on.is_empty() + } + pub(crate) fn unreachable() -> Self { Self::blocked("source_unreachable") } } +/// Pause as the pump last observed it +pub(crate) enum PauseState { + Running, + Paused { + /// Consumed then received, frozen when the pause was observed + frontier: (u64, u64), + /// Pause predates this process, so an operator may hold an older pair + refrozen: bool, + /// Last promotion gate read + polled_at: Option, + }, +} + +impl PauseState { + pub(crate) fn frontier(&self) -> Option<(u64, u64)> { + let Self::Paused { frontier, .. } = self else { + return None; + }; + Some(*frontier) + } + + pub(crate) fn refrozen(&self) -> bool { + matches!(self, Self::Paused { refrozen: true, .. }) + } +} + /// How often the gate is re-read while paused, and how long one read may take /// before the endpoint counts as unreachable. The pump publishes every tick, so /// a target that stops answering must not stall the loop with it. @@ -109,7 +140,6 @@ pub(crate) async fn promotion_gate(feed: &mut SourceFeed, pause_received: u64) - "" }; PromotionGate { - ready: blocked_on.is_empty(), blocked_on, in_recovery, replay_lsn, @@ -167,9 +197,6 @@ pub(crate) fn resume_manifest( /// fork is still in flight — the floor's contract is that a restart from it /// loses nothing, not that the natural terms have caught up to it /// (architecture/recovery.md). -/// -/// Publishes to the pruners only after the persist, the same order the status -/// loop uses: a cut must never sit above what a crash-now restart replays from. pub(crate) async fn commit_fork_resume( spill_dir: &Path, identity: &manifest::SourceIdentity, @@ -182,7 +209,7 @@ pub(crate) async fn commit_fork_resume( resume_safe: lsn.emitter_ack, filter_durable: lsn.filter_durable, published: resume_floor.get(), - fork: Some(resume.floor), + fork: Some(resume.floor()), ..manifest::FloorInputs::default() } .floor(); @@ -191,27 +218,27 @@ pub(crate) async fn commit_fork_resume( floor, source: manifest::SourceIdentity { system_id: identity.system_id, - timeline: resume.timeline, + timeline: resume.timeline(), // The fork is where the descendant begins, so the next boot can // refuse a sibling that shares its number - timeline_begin: resume.switch_lsn, + timeline_begin: resume.switch_lsn(), }, wal: manifest::WalBranch { - stream_timeline: resume.timeline, + stream_timeline: resume.timeline(), }, lsn, }; - manifest::write(spill_dir, &committed) + let persisted = manifest::write(spill_dir, &committed) .await .context("write resume manifest at the fork")?; // Descendant floor starts new position space - resume_floor.rebase(floor); - gc_floor.rebase(floor); + resume_floor.rebase(persisted); + gc_floor.rebase(persisted); tracing::info!( target: "walshadow", - timeline = resume.timeline, + timeline = resume.timeline(), floor = %floor, - switch_lsn = %resume.switch_lsn, + switch_lsn = %resume.switch_lsn(), "committed the fork resume position", ); Ok(()) @@ -733,7 +760,7 @@ mod tests { #[test] fn promotion_gate_defaults_are_not_ready() { - assert!(!PromotionGate::default().ready); + assert!(!PromotionGate::blocked("not_paused").ready()); assert_eq!( PromotionGate::blocked("not_paused").blocked_on, "not_paused" diff --git a/src/catalog/shadow_catalog.rs b/src/catalog/shadow_catalog.rs index 459a31ab..87ff8c65 100644 --- a/src/catalog/shadow_catalog.rs +++ b/src/catalog/shadow_catalog.rs @@ -62,6 +62,27 @@ pub enum CatalogError { pub type Result = std::result::Result; +/// Shadow replay parked at one position with every successor byte withheld, +/// the precondition for pinned catalog reads +#[derive(Debug)] +pub struct ParkedAt(u64); + +impl ParkedAt { + pub(crate) const fn new(lsn: u64) -> Self { + Self(lsn) + } + + pub const fn lsn(&self) -> u64 { + self.0 + } + + /// Integration tests stand in for a hold they do not run + #[cfg(feature = "test-support")] + pub const fn assume_for_test(lsn: u64) -> Self { + Self(lsn) + } +} + #[derive(Debug, Clone)] pub struct ShadowCatalogConfig { /// `pg_last_wal_replay_lsn()` poll interval @@ -331,19 +352,19 @@ impl ShadowCatalog { self.fetch_committed(Scope::Eligible, None).await } - /// [`fetch_descriptors_batch`](Self::fetch_descriptors_batch) for a caller - /// that has parked replay at `boundary` and withheld every successor byte. + /// [`fetch_descriptors_batch`](Self::fetch_descriptors_batch) pinned at + /// a [`ParkedAt`] position. /// - /// Saying so is what keeps the read out of the deadlock: the worker checks + /// Parking is what keeps the read out of the deadlock: the worker checks /// the position on both sides and may then read a catalog whose lock replay /// is holding, and no step falls back to ordinary SQL, whose parse would /// queue behind that same lock. pub async fn fetch_descriptors_batch_at( &mut self, oids: &[Oid], - boundary: u64, + at: &ParkedAt, ) -> Result<(u64, Vec)> { - self.fetch_committed(Scope::Oids(oids), Some(boundary)) + self.fetch_committed(Scope::Oids(oids), Some(at.lsn())) .await } @@ -351,9 +372,9 @@ impl ShadowCatalog { /// [`fetch_descriptors_batch_at`](Self::fetch_descriptors_batch_at). pub async fn fetch_all_descriptors_at( &mut self, - boundary: u64, + at: &ParkedAt, ) -> Result<(u64, Vec)> { - self.fetch_committed(Scope::Eligible, Some(boundary)).await + self.fetch_committed(Scope::Eligible, Some(at.lsn())).await } /// Committed catalog at one replay position. @@ -395,9 +416,8 @@ impl ShadowCatalog { } /// Descriptors as transaction `top_xid` sees them, read off shadow's pages - /// at `boundary` — the LSN the caller parked replay at. Rows the - /// transaction wrote and has not committed are included; rows it deleted - /// are not. + /// where replay is parked. Rows the transaction wrote and has not + /// committed are included; rows it deleted are not. /// /// Oids absent from `pg_class` are absent from the result, as in /// [`Self::fetch_descriptors_batch`]. @@ -405,12 +425,12 @@ impl ShadowCatalog { &mut self, oids: &[Oid], top_xid: u32, - boundary: u64, + at: &ParkedAt, ) -> Result> { let bridge = self.bridge.clone(); self.stats.fetches += 1; let rows = self - .scan_rows(&bridge, Scope::Oids(oids), top_xid, Some(boundary)) + .scan_rows(&bridge, Scope::Oids(oids), top_xid, Some(at.lsn())) .await?; let db_node = self.current_db_oid().await?; let default_tablespace = self.default_tablespace_oid().await?; diff --git a/src/ch.rs b/src/ch.rs index 5668aeb3..ffcdaecb 100644 --- a/src/ch.rs +++ b/src/ch.rs @@ -141,10 +141,15 @@ pub async fn connect_client( } } -pub async fn drain_to_end_of_stream(client: &mut BoxedAsyncClient) -> Result<(), EmitterError> { +/// Server finished the query; an INSERT's rows are durable +pub struct EndOfStream(()); + +pub async fn drain_to_end_of_stream( + client: &mut BoxedAsyncClient, +) -> Result { loop { match client.recv_event().await? { - Event::EndOfStream => return Ok(()), + Event::EndOfStream => return Ok(EndOfStream(())), Event::Exception(exc) => { return Err(EmitterError::ServerException { code: exc.code(), @@ -176,7 +181,8 @@ pub async fn exec_drain( ) -> Result<(), EmitterError> { with_timeout(timeout, async { client.send_query(sql, None).await?; - drain_to_end_of_stream(client).await + drain_to_end_of_stream(client).await?; + Ok(()) }) .await } @@ -224,21 +230,8 @@ impl ChConn { } pub async fn ready(&mut self) -> Result<&mut BoxedAsyncClient, EmitterError> { - let config = self.dest.current(); - let moved = self - .dialed - .as_ref() - .is_none_or(|dialed| !Arc::ptr_eq(dialed, &config)); - if moved || self.last_used.elapsed() >= config.idle_reconnect() { - self.client = None; - } - if self.client.is_none() { - self.client = Some(connect_client(&*config).await?); - self.dialed = Some(config); - self.last_used = Instant::now(); - self.dials += 1; - } - Ok(self.client.as_mut().expect("just connected")) + let client = self.take_ready().await?; + Ok(self.client.insert(client)) } /// Dials since the last call, excluding the one at construction @@ -305,8 +298,22 @@ impl ChConn { } async fn take_ready(&mut self) -> Result { - self.ready().await?; - Ok(self.client.take().expect("just connected")) + let config = self.dest.current(); + let moved = self + .dialed + .as_ref() + .is_none_or(|dialed| !Arc::ptr_eq(dialed, &config)); + if moved || self.last_used.elapsed() >= config.idle_reconnect() { + self.client = None; + } + if let Some(client) = self.client.take() { + return Ok(client); + } + let client = connect_client(&*config).await?; + self.dialed = Some(config); + self.last_used = Instant::now(); + self.dials += 1; + Ok(client) } } diff --git a/src/config.rs b/src/config.rs index 9628f5c1..3e29a860 100644 --- a/src/config.rs +++ b/src/config.rs @@ -334,24 +334,36 @@ pub struct ConfigResolver { /// exclusions applied. opt_in_total: AtomicU64, opt_out_total: AtomicU64, - /// TOAST heaps shadow can replay, set after filter loads replay eligibility - /// Unset outside shadow mode, which reads values from destination mirror - shadow_toast: std::sync::OnceLock, + shadow_toast: Option, +} + +/// State fixed before the pump consumes WAL +#[derive(Default)] +pub struct ResolverBoot { + /// Source PG's config tables (`SELECT *` seed, §7). Later writes arrive + /// off the WAL stream, so no seed may replace the overlay after boot + pub overlay: ConfigOverlay, + /// TOAST heaps shadow can replay. `None` outside shadow mode, which reads + /// values from destination mirror + pub shadow_toast: Option, } impl ConfigResolver { /// Build from the boot-parsed [`EmitterConfig`] plus the CLI overlay. /// Returns the shared resolver and a receiver seeded with the initial - /// (overlay-empty) snapshot; call [`seed_overlay`](Self::seed_overlay) - /// before pump start to fold in the source-PG rows. + /// snapshot, `boot.overlay` folded in. pub fn new( base: &EmitterConfig, cli: CliOverrides, toml_path: Option, cli_base: toml::Table, mapping: MappingHandle, + boot: ResolverBoot, ) -> (Arc, watch::Receiver>) { - let overlay = ConfigOverlay::default(); + let ResolverBoot { + overlay, + shadow_toast, + } = boot; let opt_in = OptInState::default(); let (initial, _) = Self::resolve(base, &overlay, &cli, &opt_in, &ColumnRules::default()); let (tx, rx) = watch::channel(Arc::new(initial)); @@ -371,19 +383,14 @@ impl ConfigResolver { pending_decl: AtomicU64::new(0), opt_in_total: AtomicU64::new(0), opt_out_total: AtomicU64::new(0), - shadow_toast: std::sync::OnceLock::new(), + shadow_toast, }); (this, rx) } - /// Set shadow TOAST eligibility once, before first opt-in - pub fn bind_shadow_toast(&self, held: ShadowHeld) { - let _ = self.shadow_toast.set(held); - } - /// `None` unless `[toast] mode = "shadow"` pub fn shadow_toast(&self) -> Option<&ShadowHeld> { - self.shadow_toast.get() + self.shadow_toast.as_ref() } /// Another receiver on the same channel. @@ -420,13 +427,6 @@ impl ConfigResolver { self.opt_out_total.load(Ordering::Relaxed) } - /// Replace the overlay wholesale (boot `SELECT *` seed, §7) and republish. - pub async fn seed_overlay(&self, overlay: ConfigOverlay) { - let mut inner = self.inner.lock().await; - inner.overlay = overlay; - self.republish(&inner).await; - } - /// Apply one WAL-driven config event at its commit LSN (§6). Mutates the /// overlay, writes the routing map under the fence, then republishes. /// Called from the reorder coordinator's barrier apply, so it runs after @@ -1514,6 +1514,7 @@ mod tests { None, toml::Table::new(), mapping.clone(), + ResolverBoot::default(), ); resolver .materialize_opt_in(&rel_desc("public", "events"), None, None) @@ -1543,6 +1544,7 @@ mod tests { None, toml::Table::new(), mapping.clone(), + ResolverBoot::default(), ); let rel = RelName::new("public", "events"); assert!(rx.borrow().tables.contains_key(&rel)); @@ -1577,6 +1579,7 @@ mod tests { None, toml::Table::new(), mapping.clone(), + ResolverBoot::default(), ); let rel = RelName::new("public", "auto"); resolver @@ -1615,6 +1618,7 @@ mod tests { None, toml::Table::new(), mapping.clone(), + ResolverBoot::default(), ); let rel = RelName::new("public", "events"); resolver @@ -1656,6 +1660,7 @@ mod tests { None, toml::Table::new(), mapping.clone(), + ResolverBoot::default(), ); let mut desc = rel_desc("public", "events"); desc.attributes.push(RelAttr { @@ -1927,6 +1932,7 @@ mod tests { None, toml::Table::new(), mapping, + ResolverBoot::default(), ); let upsert = |ty: &str| ConfigEvent::ColumnUpserted { rel: RelName::new("public", "t"), @@ -1980,6 +1986,7 @@ mod tests { None, toml::Table::new(), mapping, + ResolverBoot::default(), ); let rel = RelName::new("app", "later"); resolver @@ -1995,15 +2002,6 @@ mod tests { async fn seed_and_apply_republish() { let base = base_with("retain"); let mapping = dummy_handles(); - let (resolver, mut rx) = ConfigResolver::new( - &base, - CliOverrides::default(), - None, - toml::Table::new(), - mapping, - ); - assert_eq!(rx.borrow().drop_table_strategy, DropTableStrategy::Retain); - let overlay = ConfigOverlay { global: Some(GlobalRow { drop_table_strategy: Some("drop".into()), @@ -2011,8 +2009,17 @@ mod tests { }), ..Default::default() }; - resolver.seed_overlay(overlay).await; - assert!(rx.changed().await.is_ok()); + let (resolver, mut rx) = ConfigResolver::new( + &base, + CliOverrides::default(), + None, + toml::Table::new(), + mapping, + ResolverBoot { + overlay, + ..Default::default() + }, + ); assert_eq!( rx.borrow_and_update().drop_table_strategy, DropTableStrategy::Drop @@ -2038,6 +2045,7 @@ mod tests { None, toml::Table::new(), mapping, + ResolverBoot::default(), ); resolver.reload().await.unwrap(); assert_eq!(rx.borrow().drop_table_strategy, DropTableStrategy::Retain); @@ -2162,6 +2170,7 @@ mod tests { Some(path.clone()), toml::Table::new(), dummy_handles(), + ResolverBoot::default(), ); assert_eq!(rx.borrow_and_update().source.host, "pg-a"); @@ -2189,6 +2198,7 @@ mod tests { Some(path.clone()), toml::Table::new(), dummy_handles(), + ResolverBoot::default(), ); let billing_rels = |rx: &mut watch::Receiver>| { let snap = rx.borrow_and_update(); @@ -2267,6 +2277,7 @@ mod tests { Some(path.clone()), toml::Table::new(), dummy_handles(), + ResolverBoot::default(), ); let held = resolver.inner.lock().await; let queued = tokio::spawn({ diff --git a/src/decode/visibility.rs b/src/decode/visibility.rs index 1efba52a..d6707435 100644 --- a/src/decode/visibility.rs +++ b/src/decode/visibility.rs @@ -349,10 +349,11 @@ impl PgXactPatch { self.aborted.extend(subxids); } - /// Run-length encode the ascending stretches once harvesting is done - pub fn seal(&mut self) { + /// End harvesting, run-length encoding the ascending stretches + pub fn seal(mut self) -> SealedPatch { self.committed.optimize(); self.aborted.optimize(); + SealedPatch(self) } pub fn len(&self) -> usize { @@ -364,6 +365,20 @@ impl PgXactPatch { } } +/// Harvested [`PgXactPatch`], the only form xid resolution reads +#[derive(Debug, Default)] +pub struct SealedPatch(PgXactPatch); + +impl SealedPatch { + pub fn len(&self) -> usize { + self.0.len() + } + + pub fn is_empty(&self) -> bool { + self.0.is_empty() + } +} + /// Patch-over-accum xid resolution, plus optional pg_multixact for /// `HEAP_XMAX_IS_MULTI` xmax. pub struct PgXactView<'a> { @@ -373,10 +388,10 @@ pub struct PgXactView<'a> { } impl<'a> PgXactView<'a> { - pub fn new(accum: &'a PgXactAccum, patch: &'a PgXactPatch) -> Self { + pub fn new(accum: &'a PgXactAccum, patch: &'a SealedPatch) -> Self { Self { accum, - patch, + patch: &patch.0, multi: None, } } @@ -701,6 +716,7 @@ mod tests { let mut p = PgXactPatch::new(); p.commit(100, &[101, 102]); p.abort(200, &[]); + let p = p.seal(); let v = PgXactView::new(&a, &p); assert_eq!(v.xid_status(100), XidStatus::Committed); assert_eq!(v.xid_status(102), XidStatus::Committed, "subxid patched"); @@ -767,7 +783,7 @@ mod tests { (200, TRANSACTION_STATUS_COMMITTED), ], ); - let p = PgXactPatch::new(); + let p = PgXactPatch::new().seal(); let v = PgXactView::new(&a, &p); // xmin committed via pg_xact, no xmax assert_eq!(tuple_visibility(100, 0, 0, Some(&v)), Visibility::Emit); @@ -810,6 +826,7 @@ mod tests { let a = accum_with(0, &[]); let mut p = PgXactPatch::new(); p.commit(500, &[]); + let p = p.seal(); let v = PgXactView::new(&a, &p); assert_eq!(tuple_visibility(500, 0, 0, Some(&v)), Visibility::Emit); // Same for a gap-committed deleter: tuple is dead @@ -922,7 +939,7 @@ mod tests { (911, TRANSACTION_STATUS_ABORTED), ], ); - let p = PgXactPatch::new(); + let p = PgXactPatch::new().seal(); let m = mx_accum( &[(10, 100), (11, 102), (20, 102), (21, 104), (30, 200)], &[ @@ -959,6 +976,7 @@ mod tests { // Gap-patch-committed updater: dead let mut p3 = PgXactPatch::new(); p3.commit(950, &[]); + let p3 = p3.seal(); let v3 = PgXactView::new(&a, &p3).with_multixact(&m2); assert_eq!(tuple_visibility(100, 10, mask, Some(&v3)), Visibility::Skip); // View without pg_multixact: unresolvable, caller aborts @@ -1030,7 +1048,7 @@ mod tests { let accum = read_pg_xact(dir).await.unwrap(); // No pg_multixact/ at all: a cluster that never made one let multi = read_pg_multixact(dir, 17).await.unwrap(); - let patch = PgXactPatch::new(); + let patch = PgXactPatch::new().seal(); let view = PgXactView::new(&accum, &patch).with_multixact(&multi); assert_eq!(view.xid_status(700), XidStatus::Committed); diff --git a/src/emit/pipeline/ack.rs b/src/emit/pipeline/ack.rs index 90dd8f51..87e5e38e 100644 --- a/src/emit/pipeline/ack.rs +++ b/src/emit/pipeline/ack.rs @@ -6,10 +6,16 @@ //! watermark advances through contiguous done seqs only //! //! One commit may span several seqs. Only the final seq publishes its -//! `commit_lsn`, earlier slices gate contiguity via -//! [`AckHandle::register_partial`] +//! `commit_lsn`, earlier slices gate contiguity via [`Publish::Partial`] //! -//! Inserter sends [`AckEvent::Acked`] only after draining `EndOfStream` +//! [`SeqAlloc::open`] is the only registration path. An allocator and its +//! seqs send only to the collector that made them, and [`OpenSeq::place`] +//! consumes its seq. Types do not stop two allocators overlapping or a seq +//! dropping unplaced: collector fails on a repeat registration, and an +//! unplaced seq pins the frontier where [`AckSnapshot::stall_reason`] names it +//! +//! [`AckHandle::acked`] takes the [`EndOfStream`] proof, so only a drained +//! INSERT acks use std::collections::BTreeMap; use std::sync::Arc; @@ -17,6 +23,7 @@ use std::sync::Arc; use tokio::sync::{mpsc, watch}; use tokio::task::JoinHandle; +use crate::ch::EndOfStream; use crate::emit::pipeline::Fatal; use crate::pos::{AckFrontier, EmitterAck, Gate, GateClosed, Monotone, Pos}; @@ -300,31 +307,17 @@ pub struct AckHandle { } impl AckHandle { - /// Register a commit's final (or only) seq; its `commit_lsn` publishes - /// once the contiguous-done frontier passes it. - pub fn register(&self, seq: u64, commit_lsn: u64) { - let _ = self.tx.send(AckEvent::Register { - seq, - commit_lsn, - publish: true, - }); - } - - /// Register a non-final slice of a multi-seq commit: counts toward - /// contiguity but never advances `emitter_ack` (see module doc). - pub fn register_partial(&self, seq: u64, commit_lsn: u64) { - let _ = self.tx.send(AckEvent::Register { - seq, - commit_lsn, - publish: false, - }); + /// Rows ClickHouse confirmed durable + pub fn acked(&self, _: EndOfStream, counts: Vec<(u64, u64)>) { + self.send_acked(counts); } - pub fn placed(&self, seq: u64, rows: u64) { - let _ = self.tx.send(AckEvent::Placed { seq, rows }); + /// Null tail: rows dropped by design count as done + pub(super) fn swallowed(&self, counts: Vec<(u64, u64)>) { + self.send_acked(counts); } - pub fn acked(&self, counts: Vec<(u64, u64)>) { + fn send_acked(&self, counts: Vec<(u64, u64)>) { if !counts.is_empty() { let _ = self.tx.send(AckEvent::Acked { counts }); } @@ -346,6 +339,78 @@ impl AckHandle { pub async fn wait_through(&self, seq: u64) -> Result<(), GateClosed> { self.frontier.wait(Pos::new(seq)).await.map(|_| ()) } + + /// Seq source registering on this collector, opening at `first` + pub fn seqs(&self, first: u64) -> SeqAlloc { + SeqAlloc { + tx: self.tx.clone(), + next: first, + } + } +} + +/// Whether a done seq advances `emitter_ack` to its `commit_lsn` +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum Publish { + /// Commit's final (or only) seq + Commit, + /// Non-final slice of a multi-seq commit, gates contiguity only + Partial, +} + +/// Dense seq source for one producer; registers in order without gaps +#[derive(Debug)] +pub struct SeqAlloc { + tx: mpsc::UnboundedSender, + next: u64, +} + +impl SeqAlloc { + /// One past the last opened seq + pub const fn next(&self) -> u64 { + self.next + } + + pub fn open(&mut self, commit_lsn: u64, publish: Publish) -> OpenSeq { + let seq = self.next; + self.next += 1; + let _ = self.tx.send(AckEvent::Register { + seq, + commit_lsn, + publish: publish == Publish::Commit, + }); + OpenSeq { + tx: self.tx.clone(), + seq, + } + } + + /// Ordering marker carrying no rows + pub fn mark(&mut self, commit_lsn: u64) { + self.open(commit_lsn, Publish::Commit).place(0); + } +} + +/// Registered seq awaiting its row count +#[must_use = "an unplaced seq pins the ack frontier"] +#[derive(Debug)] +pub struct OpenSeq { + tx: mpsc::UnboundedSender, + seq: u64, +} + +impl OpenSeq { + pub const fn seq(&self) -> u64 { + self.seq + } + + /// Every row of this seq must already be on the batcher channel + pub fn place(self, rows: u64) { + let _ = self.tx.send(AckEvent::Placed { + seq: self.seq, + rows, + }); + } } /// Spawn the collector actor. When all [`AckHandle`] clones drop it drains diff --git a/src/emit/pipeline/bootstrap.rs b/src/emit/pipeline/bootstrap.rs index d38b36d4..6aeeb850 100644 --- a/src/emit/pipeline/bootstrap.rs +++ b/src/emit/pipeline/bootstrap.rs @@ -8,14 +8,15 @@ use std::sync::Arc; use std::sync::atomic::{AtomicU64, Ordering}; use tokio::sync::mpsc; +use walrus::pg::walparser::RelFileNode; use crate::backfill::backup_page_walk::{BackfillTuple, CatalogMap}; -use crate::backfill::spool::{DeferredReader, DeferredSpool, SpoolMark}; +use crate::backfill::spool::{DeferredReader, DeferredSpool, Replayable, SpoolMark}; use crate::backfill::walk_barrier::{WALK_CHECKPOINT_PERIOD, WalkBarrier}; use crate::config::ResolvedConfig; use crate::decode::heap_decoder::{ColumnValue, ToastPointer}; use crate::emit::ch_emitter::EmitterStats; -use crate::emit::pipeline::ack::AckHandle; +use crate::emit::pipeline::ack::{AckHandle, OpenSeq, Publish, SeqAlloc}; use crate::emit::pipeline::batcher::{BatcherMsg, RoutedRow, RowChunk}; use crate::emit::pipeline::decode::DECODE_CHUNK_BYTES; use crate::emit::route::{RouteSnapshot, RowPolicy, freeze_routes}; @@ -66,7 +67,7 @@ impl<'a> DeferredFootprint<'a> { } /// Take over bytes a lane published, so replay releases them - fn adopt(stats: &'a EmitterStats, spool: &DeferredSpool) -> Self { + fn adopt(stats: &'a EmitterStats, spool: &Replayable) -> Self { Self { stats, resident: spool.resident_bytes() as u64, @@ -187,9 +188,9 @@ pub async fn drain( }; let routes = freeze_routes(&mapping, config.as_deref(), &row_policy); let mut footprint = DeferredFootprint::new(&stats); - let mut next_seq = 0; + let mut seqs = ack.seqs(0); let mut rows_routed = 0; - let mut open = None; + let mut open: Option<(RelFileNode, OpenSeq, u64)> = None; let mut chunk_batch = Vec::new(); let mut chunk_batch_bytes = 0; let mut out = RowBuf::default(); @@ -201,20 +202,20 @@ pub async fn drain( let rfn = tuple.rfn; let source_lsn = tuple.source_lsn; - let same = matches!(&open, Some((r, _, _)) if *r == rfn); - let seq = if same { - open.as_ref().expect("same implies open").1 + let seq = if let Some((r, seq, _)) = &open + && *r == rfn + { + seq.seq() } else { - if let Some((_, prev_seq, prev_rows)) = open.take() { + if let Some((_, prev, prev_rows)) = open.take() { // Every row of the closing seq on the channel before its // expected count is published out.flush(&msg_tx).await?; - ack.placed(prev_seq, prev_rows); + prev.place(prev_rows); } - let s = next_seq; - next_seq += 1; - ack.register(s, source_lsn); - open = Some((rfn, s, 0)); + let seq = seqs.open(source_lsn, Publish::Commit); + let s = seq.seq(); + open = Some((rfn, seq, 0)); s }; @@ -278,7 +279,7 @@ pub async fn drain( } out.flush(&msg_tx).await?; if let Some((_, seq, rows)) = open.take() { - ack.placed(seq, rows); + seq.place(rows); } let mut mark = SpoolMark::default(); if let Some(spool) = deferred.as_mut() { @@ -288,13 +289,14 @@ pub async fn drain( .map_err(|e| format!("bootstrap: deferred spool checkpoint: {e}"))?; footprint.publish(spool); } - b.publish_drain(consumed, next_seq, mark).await; + b.publish_drain(consumed, seqs.next(), mark).await; } } out.flush(&msg_tx).await?; if let Some((_, seq, rows)) = open.take() { - ack.placed(seq, rows); + seq.place(rows); } + let next_seq = seqs.next(); if !chunk_batch.is_empty() { flush_chunks(&resolver, &mut chunk_batch).await?; @@ -312,7 +314,15 @@ pub async fn drain( Some(spool) => { footprint.hand_off(); let resolved = resolve_spooled( - spool, &routes, &catalog, &msg_tx, &ack, &stats, &resolver, next_seq, None, + spool.into(), + &routes, + &catalog, + &msg_tx, + &ack, + &stats, + &resolver, + next_seq, + None, ) .await?; Ok(BootstrapDrainOutcome { @@ -336,7 +346,7 @@ pub async fn drain( /// sibling's chunk files #[allow(clippy::too_many_arguments)] pub async fn drain_deferred( - spool: DeferredSpool, + spool: Replayable, catalog: &CatalogMap, mapping: &MappingSnapshot, msg_tx: &mpsc::Sender, @@ -365,7 +375,7 @@ pub struct ReplayCheckpoint<'a> { /// row that routes so an all-unmapped spool leaves no seq to prove #[allow(clippy::too_many_arguments)] async fn resolve_spooled( - spool: DeferredSpool, + spool: Replayable, routes: &ahash::HashMap>, catalog: &CatalogMap, msg_tx: &mpsc::Sender, @@ -387,7 +397,7 @@ async fn resolve_spooled( let mut seq = None; let mut placed = 0u64; let total_bytes = spool.spooled_bytes(); - let mut next_seq = first_seq; + let mut seqs = ack.seqs(first_seq); let mut total_placed = 0; let mut ticker = Ticker::new(WALK_CHECKPOINT_PERIOD); let mut replay = spool @@ -403,28 +413,19 @@ async fn resolve_spooled( let remaining = replay.remaining_file_bytes(); let (next, routed) = tokio::join!( prepare_batch(&mut replay, routes, catalog, stats, resolver), - route_batch( - batch, - &mut out, - msg_tx, - ack, - next_seq, - &mut seq, - &mut placed - ), + route_batch(batch, &mut out, msg_tx, &mut seqs, &mut seq, &mut placed), ); routed?; progress.advance(remaining); if let Some(c) = checkpoint.as_mut() { out.flush(msg_tx).await?; if let Some(s) = seq.take() { - ack.placed(s, placed); - next_seq = s + 1; + s.place(placed); } total_placed += placed; placed = 0; if ticker.fire() { - c.tail.checkpoint(next_seq).await?; + c.tail.checkpoint(seqs.next()).await?; c.state.offset = total_bytes - remaining; c.state.rows += total_placed; total_placed = 0; @@ -436,7 +437,7 @@ async fn resolve_spooled( out.flush(msg_tx).await?; let rows_routed = match checkpoint.as_mut() { Some(c) => { - c.tail.checkpoint(next_seq).await?; + c.tail.checkpoint(seqs.next()).await?; c.state.offset = total_bytes; c.state.rows += total_placed; c.state.save(c.dir).await.map_err(|e| e.to_string())?; @@ -451,21 +452,17 @@ async fn resolve_spooled( placed } }; - let next_seq = match seq { - Some(s) => { - ack.placed(s, placed); - s + 1 - } - None => next_seq, - }; + if let Some(s) = seq { + s.place(placed); + } Ok(BootstrapDrainOutcome { - next_seq, + next_seq: seqs.next(), rows_routed, deferred: None, }) } -fn bump(open: &mut Option<(walrus::pg::walparser::RelFileNode, u64, u64)>, rows_routed: &mut u64) { +fn bump(open: &mut Option<(RelFileNode, OpenSeq, u64)>, rows_routed: &mut u64) { if let Some(slot) = open.as_mut() { slot.2 += 1; } @@ -617,18 +614,16 @@ async fn route_batch( batch: ResolvedReplayBatch, out: &mut RowBuf, msg_tx: &mpsc::Sender, - ack: &AckHandle, - first_seq: u64, - seq: &mut Option, + seqs: &mut SeqAlloc, + seq: &mut Option, placed: &mut u64, ) -> Result<(), String> { let ResolvedReplayBatch { rows, permit } = batch; for row in rows { let mut tuple = row.tuple; - let at = *seq.get_or_insert_with(|| { - ack.register(first_seq, tuple.source_lsn); - first_seq - }); + let at = seq + .get_or_insert_with(|| seqs.open(tuple.source_lsn, Publish::Commit)) + .seq(); render_ext_columns(&row.rel.attributes, &mut tuple.columns); out.push(msg_tx, at, row.rel, row.route, tuple, permit.clone()) .await?; @@ -1593,7 +1588,7 @@ mod tests { ); let resolved = drain_deferred( - spool, + spool.into(), &catalog, &mapping, &msg_tx, @@ -1682,7 +1677,7 @@ mod tests { for (spool, ack, collector, first_seq) in lanes { remaining -= spool.resident_bytes() as u64; drain_deferred( - spool, + spool.into(), &catalog, &mapping, &msg_tx, @@ -1896,7 +1891,7 @@ mod tests { .bootstrap_deferred_spool_bytes .store(total, Ordering::Relaxed); let result = drain_deferred( - spool, + spool.into(), &catalog, &mapping, &tail.msg_tx, @@ -1928,7 +1923,7 @@ mod tests { .store(total, Ordering::Relaxed); let tail = OwnedTail::null(); let result = drain_deferred( - spool, + spool.into(), &catalog, &mapping, &tail.msg_tx, diff --git a/src/emit/pipeline/inserter.rs b/src/emit/pipeline/inserter.rs index 76c660f4..b8be66a9 100644 --- a/src/emit/pipeline/inserter.rs +++ b/src/emit/pipeline/inserter.rs @@ -7,15 +7,15 @@ //! Native block over the batch's owned slabs, and runs one `send_query` + //! `send_data` + `send_data_end` + drain-to-`EndOfStream` INSERT. //! -//! Durability invariant: [`AckHandle::acked`] fires **only after** the drain -//! returns. Until then a connection drop replays the still-owned batch (CH +//! Durability invariant: [`AckHandle::acked`] takes the drain's +//! [`EndOfStream`] proof. Until then a connection drop replays the still-owned batch (CH //! dedups by `_lsn`). Retry-exhaustion is fatal: the watermark can't advance //! without this batch. use clickhouse_c::{Allocator, BlockBuilder, ColumnBuilder, TypeAst}; use tokio::task::JoinHandle; -use crate::ch::{ChConn, EmitterError, drain_to_end_of_stream, with_timeout}; +use crate::ch::{ChConn, EmitterError, EndOfStream, drain_to_end_of_stream, with_timeout}; use crate::config::DestEmitter; use crate::emit::ch_emitter::{EmitterStats, build_leaf, build_root}; use crate::emit::pipeline::Fatal; @@ -39,20 +39,19 @@ struct Inserter { } impl Inserter { - fn ensure_asts(&mut self, meta: &BatchMeta) -> Result<(), EmitterError> { - let fresh = self - .asts - .get(&meta.table_key) - .is_none_or(|(epoch, _)| *epoch != meta.schema_epoch); - if fresh { - let mut parsed = Vec::with_capacity(meta.columns.len()); - for col in &meta.columns { - parsed.push(TypeAst::parse(&col.type_repr, self.alloc)?); - } - self.asts - .insert(meta.table_key.clone(), (meta.schema_epoch, parsed)); + /// Owned out of the cache so no `&[TypeAst]` lives across the send + /// await; the caller puts them back + fn take_asts(&mut self, meta: &BatchMeta) -> Result<(u64, Vec), EmitterError> { + if let Some((epoch, asts)) = self.asts.remove(&meta.table_key) + && epoch == meta.schema_epoch + { + return Ok((epoch, asts)); } - Ok(()) + let mut parsed = Vec::with_capacity(meta.columns.len()); + for col in &meta.columns { + parsed.push(TypeAst::parse(&col.type_repr, self.alloc)?); + } + Ok((meta.schema_epoch, parsed)) } /// Bounded reconnect+retry around one prepared INSERT. Only `bb` @@ -63,7 +62,7 @@ impl Inserter { &mut self, sql: &str, bb: &BlockBuilder<'_>, - ) -> Result<(), EmitterError> { + ) -> Result { let insert_timeout = self.client.config().insert_timeout; let started = std::time::Instant::now(); let result = self @@ -74,7 +73,6 @@ impl Inserter { client.send_query(sql, None).await?; client.send_data(Some(bb)).await?; client.send_data_end().await?; - // Only after EndOfStream returns are rows durable and ackable drain_to_end_of_stream(&mut client).await }) .await; @@ -98,16 +96,13 @@ impl Inserter { async fn run(mut self, rx: async_channel::Receiver, fatal: Fatal) { while let Ok(ResolvedBatch { batch, resolved }) = rx.recv().await { - if let Err(e) = self.ensure_asts(&batch.meta) { - fatal.set(format!("inserter type parse: {e}")); - break; - } - // Own the asts (`Vec` is `Send`); index inline so no - // `&[TypeAst]` binding lives across the send await - let (epoch, asts) = self - .asts - .remove(&batch.meta.table_key) - .expect("ensure_asts inserted"); + let (epoch, asts) = match self.take_asts(&batch.meta) { + Ok(asts) => asts, + Err(e) => { + fatal.set(format!("inserter type parse: {e}")); + break; + } + }; let result = 'send: { let encode_started = std::time::Instant::now(); let leaves: Vec>> = match batch @@ -165,7 +160,7 @@ impl Inserter { self.asts .insert(batch.meta.table_key.clone(), (epoch, asts)); match result { - Ok(()) => { + Ok(durable) => { self.stats .rows_emitted .fetch_add(batch.n_rows as u64, Ordering::Relaxed); @@ -181,7 +176,7 @@ impl Inserter { self.stats .inserter_batches_in .fetch_add(1, Ordering::Relaxed); - self.ack.acked(batch.per_seq); + self.ack.acked(durable, batch.per_seq); } Err(e) => { fatal.set(format!("inserter: {e}")); diff --git a/src/emit/pipeline/mod.rs b/src/emit/pipeline/mod.rs index e6915bf4..57e18477 100644 --- a/src/emit/pipeline/mod.rs +++ b/src/emit/pipeline/mod.rs @@ -124,10 +124,10 @@ pub struct PipelineConfig { /// A database with no entry streams from the opt-in LSN only pub backfillers: HashMap>, /// Durable queue of deferred toast-mirror retires, loaded from the - /// spill dir; entries due at resume retire via the post-spawn - /// [`reorder::ReorderSink::flush_due_retires`] call + /// spill dir; entries due at resume retire in + /// [`reorder::BootingReorder::boot`] pub retires: crate::toast::toast_retire::RetireLedger, - /// Pending backup rows; recover outcomes with [`reorder::ReorderSink::settle_pending_boot`] + /// Pending backup rows; [`reorder::BootingReorder::boot`] recovers outcomes pub pending_rows: crate::backfill::visibility_pending::SharedPendingLedger, /// Persisted resolved floor (aligned, archive-clamped), seeded at the /// resolved start; pruners cut against it verbatim @@ -168,13 +168,13 @@ impl PipelineHandle { } impl PipelineConfig { - /// Stand up the pipeline. Returns the reorder sink (drive via the daemon's - /// `QueueingRecordSink`) and a handle for shutdown / watermark reads. Fails + /// Stand up the pipeline. Returns the reorder sink to boot then drive via + /// the daemon's `QueueingRecordSink`, and a handle for shutdown / watermark reads. Fails /// only if an inserter connection can't open. pub async fn spawn( self, emitter_ack: Arc>, - ) -> Result<(reorder::ReorderSink, PipelineHandle), EmitterError> { + ) -> Result<(reorder::BootingReorder, PipelineHandle), EmitterError> { let PipelineConfig { emitter, decoder_pool_size, @@ -264,7 +264,7 @@ impl PipelineConfig { let ack_probe = ack.probe(); let plan_dir = buffer.lock().await.scratch_dir().to_path_buf(); - let reorder = reorder::ReorderSink::new( + let reorder = reorder::ReorderSink::booting( buffer, dbs, pending, diff --git a/src/emit/pipeline/plan_spool.rs b/src/emit/pipeline/plan_spool.rs index aeeb749e..f3ffcb4a 100644 --- a/src/emit/pipeline/plan_spool.rs +++ b/src/emit/pipeline/plan_spool.rs @@ -25,7 +25,7 @@ use std::sync::Arc; use thiserror::Error; -use crate::decode::heap_decoder::DescribedHeap; +use crate::decode::heap_decoder::{DescribedHeap, HeapOp}; use crate::emit::route::{RouteSnapshot, RoutedHeap}; use crate::schema::RelDescriptor; use crate::xact::spill::{self, Cursor, SpillError}; @@ -215,10 +215,17 @@ impl PlanWriter { } } - /// Mirror-row fence for the next `HeapOp::Truncate` heap: executor puts - /// rows up to `upto` before applying the truncate - pub fn note_truncate_cursor(&mut self, upto: usize) { + /// Append a `HeapOp::Truncate` heap with its mirror-row fence: executor + /// puts rows up to `upto` before applying the truncate + pub fn push_truncate( + &mut self, + heap: &DescribedHeap, + route: Option<&Arc>, + upto: usize, + ) -> Result<()> { + self.push_heap(heap, route)?; self.truncate_rows.push(upto); + Ok(()) } /// Write the seal frame and freeze the header. Seal is exempt from the @@ -349,6 +356,7 @@ impl SealedPlan { offset: 4, next_heap: 0, next_control: 0, + next_truncate: 0, done: false, }) } @@ -362,6 +370,26 @@ impl SealedPlan { while rd.next_item()?.is_some() {} Ok(()) } + + /// Mem-resident plans hold the bytes validated at write; file-backed + /// plans re-read from disk, so they walk [`Self::verify`] first + pub fn into_verified(self) -> Result { + if self.path().is_some() { + self.verify()?; + } + Ok(VerifiedPlan(self)) + } +} + +/// Plan whose bytes passed checksum, safe to start side effects from +pub struct VerifiedPlan(SealedPlan); + +impl std::ops::Deref for VerifiedPlan { + type Target = SealedPlan; + + fn deref(&self) -> &SealedPlan { + &self.0 + } } impl Drop for SealedPlan { @@ -376,6 +404,11 @@ impl Drop for SealedPlan { pub enum PlanItem<'p> { Control(&'p OrderedEvent), Heap(RoutedHeap), + /// Mirror rows below `upto` put before the truncate applies + Truncate { + heap: RoutedHeap, + upto: usize, + }, } /// Linear replay over a [`SealedPlan`]: verifies every frame checksum, @@ -387,6 +420,7 @@ pub struct PlanReader<'p> { offset: u64, next_heap: u64, next_control: usize, + next_truncate: usize, done: bool, } @@ -457,14 +491,26 @@ impl<'p> PlanReader<'p> { Some(r.clone()) }; self.next_heap += 1; - Ok(Some(PlanItem::Heap(RoutedHeap { + let heap = RoutedHeap { described: DescribedHeap { decoded, descriptor, descriptor_valid_from: valid_from, }, route, - }))) + }; + if heap.described.decoded.op != HeapOp::Truncate { + return Ok(Some(PlanItem::Heap(heap))); + } + let Some(&upto) = self.plan.truncate_rows.get(self.next_truncate) else { + return Err(format(format!( + "truncate heap {} has no mirror-row fence, header has {}", + self.next_truncate, + self.plan.truncate_rows.len(), + ))); + }; + self.next_truncate += 1; + Ok(Some(PlanItem::Truncate { heap, upto })) } Some(&TAG_SEAL) => { let count = u64::from_le_bytes( @@ -479,6 +525,13 @@ impl<'p> PlanReader<'p> { self.next_heap, self.plan.heap_count, ))); } + if self.next_truncate != self.plan.truncate_rows.len() { + return Err(format(format!( + "replayed {} truncates, header fences {}", + self.next_truncate, + self.plan.truncate_rows.len(), + ))); + } let mut probe = [0u8; 1]; if self.input.read(&mut probe)? != 0 { return Err(format("trailing bytes after seal".into())); @@ -607,7 +660,7 @@ mod tests { while let Some(item) = rd.next_item().unwrap() { order.push(match item { PlanItem::Control(c) => format!("{:?}", c.event), - PlanItem::Heap(h) => { + PlanItem::Heap(h) | PlanItem::Truncate { heap: h, .. } => { assert!( Arc::ptr_eq(&h.described.descriptor, &plan.descriptors[0].0) || Arc::ptr_eq(&h.described.descriptor, &plan.descriptors[1].0), @@ -763,6 +816,113 @@ mod tests { assert!(matches!(err, PlanSpoolError::Unsealed), "{err}"); } + /// Replay a memory plan after `tamper`, returning the first error + fn replay_err( + push: impl FnOnce(&mut PlanWriter), + tamper: impl FnOnce(&mut SealedPlan, &mut Vec), + ) -> PlanSpoolError { + let tmp = tempfile::tempdir().unwrap(); + let mut w = + PlanWriter::create(tmp.path().join("1.plan"), 1 << 20, DEFAULT_PLAN_MEM_MAX).unwrap(); + push(&mut w); + let mut plan = w.seal(0x2000, 42).unwrap(); + let mut buf = match std::mem::replace(&mut plan.store, PlanStore::Mem(Vec::new())) { + PlanStore::Mem(buf) => buf, + PlanStore::File(_) => unreachable!("small plan stays in memory"), + }; + tamper(&mut plan, &mut buf); + plan.store = PlanStore::Mem(buf); + let mut rd = match plan.replay() { + Ok(rd) => rd, + Err(e) => return e, + }; + loop { + match rd.next_item() { + Ok(Some(_)) => {} + Ok(None) => panic!("tampered plan replayed clean"), + Err(e) => return e, + } + } + } + + /// Replay cross-checks every frame against the header it was sealed with + #[test] + fn inconsistent_plan_fails_replay() { + let d = descriptor(16500); + let r = route(); + let mut truncate = heap(&d, 100, 10); + truncate.decoded.op = HeapOp::Truncate; + let one = |w: &mut PlanWriter| w.push_heap(&heap(&d, 100, 10), None).unwrap(); + let mut trailing_heap = vec![TAG_HEAP]; + trailing_heap.extend_from_slice(&0u32.to_le_bytes()); + trailing_heap.extend_from_slice(&ROUTE_NONE.to_le_bytes()); + spill::encode_heap_into(&mut trailing_heap, &heap(&d, 100, 10).decoded); + trailing_heap.push(0); + let cases: Vec<(&str, PlanSpoolError)> = vec![ + ("bad plan magic", replay_err(one, |_, b| b[0] ^= 0xFF)), + ( + "unsupported plan version", + replay_err(one, |_, b| b[2] ^= 0xFF), + ), + ("dict id", replay_err(one, |p, _| p.descriptors.clear())), + ( + "route id", + replay_err( + |w| w.push_heap(&heap(&d, 100, 10), Some(&r)).unwrap(), + |p, _| p.routes.clear(), + ), + ), + ( + "no mirror-row fence", + replay_err(|w| w.push_heap(&truncate, None).unwrap(), |_, _| {}), + ), + ("seal count", replay_err(one, |p, _| p.heap_count += 1)), + ( + "header fences", + replay_err(one, |p, _| p.truncate_rows.push(0)), + ), + ( + "trailing bytes after seal", + replay_err(one, |_, b| b.push(0)), + ), + ( + "unknown plan frame tag", + replay_err(|w| w.write_frame(&[0x7F], false).unwrap(), |_, _| {}), + ), + ( + "empty plan frame", + replay_err(|w| w.write_frame(&[], false).unwrap(), |_, _| {}), + ), + ( + "trailing bytes", + replay_err( + |w| { + one(w); + w.write_frame(&trailing_heap, false).unwrap(); + }, + |_, _| {}, + ), + ), + ( + "", + replay_err( + |w| { + one(w); + w.write_frame(&trailing_heap[..10], false).unwrap(); + }, + |_, _| {}, + ), + ), + ]; + for (want, err) in cases { + assert!( + matches!(err, PlanSpoolError::Format { .. }), + "{want}: {err}" + ); + assert!(err.to_string().contains(want), "{want}: {err}"); + } + } + /// Byte cap bounds writes; an unsealed writer unlinks its file on drop #[test] fn budget_bounds_writes_and_drop_unlinks() { diff --git a/src/emit/pipeline/planner.rs b/src/emit/pipeline/planner.rs index 45e6d892..46036ca7 100644 --- a/src/emit/pipeline/planner.rs +++ b/src/emit/pipeline/planner.rs @@ -157,10 +157,10 @@ impl<'a, V: PlanRouteView> Planner<'a, V> { self.view.apply(&e).await.map_err(PlanError::View)?; self.writer.push_control(e, self.rows_base + cursor); } - WalkStep::Truncate(heap) => { - self.writer.note_truncate_cursor(self.rows_base + cursor); + WalkStep::Truncate { heap, upto } => { let route = self.view.route_for(&heap); - self.writer.push_heap(&heap, route.as_ref())?; + self.writer + .push_truncate(&heap, route.as_ref(), self.rows_base + upto)?; } WalkStep::Heap(mut heap) => { // Route before validation and detoast: unmapped rows @@ -203,7 +203,7 @@ mod tests { use crate::xact::xact_buffer::raw_fixtures::{ inject_ordinary, int4_descriptor, multi_insert_raw, }; - use crate::xact::xact_buffer::{FailClosedReason, XactBuffer, XactBufferConfig}; + use crate::xact::xact_buffer::{FailClosedReason, StashResolved, XactBuffer, XactBufferConfig}; use ahash::{HashMap, HashMapExt, HashSet, HashSetExt}; use walrus::pg::walparser::RelFileNode; @@ -397,7 +397,10 @@ mod tests { let resolver = ToastResolver::disabled(); let mut planner = Planner::create(tmp.path().join("1.plan"), 1 << 20, &mut view, &resolver).unwrap(); - let mut drain = b.drain_committed(1, 42, 0x2000, &[], false).await.unwrap(); + let mut drain = b + .drain_committed(StashResolved::nothing_stashed(1), 42, 0x2000, &[], false) + .await + .unwrap(); // 1-row slices force the multi-batch path while let Some(batch) = drain.next_batch(1, usize::MAX, None).await.unwrap() { let is_final = batch.is_final; @@ -416,7 +419,7 @@ mod tests { while let Some(item) = rd.next_item().unwrap() { order.push(match item { PlanItem::Control(_) => "e".to_string(), - PlanItem::Heap(h) => format!( + PlanItem::Heap(h) | PlanItem::Truncate { heap: h, .. } => format!( "h{}{}", h.described.decoded.source_lsn, if h.route.is_some() { "r" } else { "-" } @@ -466,7 +469,10 @@ mod tests { ); let path = tmp.path().join("1.plan"); let mut planner = Planner::create(path.clone(), 1 << 20, &mut view, &resolver).unwrap(); - let mut drain = b.drain_committed(1, 42, 0x2000, &[], false).await.unwrap(); + let mut drain = b + .drain_committed(StashResolved::nothing_stashed(1), 42, 0x2000, &[], false) + .await + .unwrap(); // 1-row slices: the valid row plans before the toast row fails let first = drain .next_batch(1, usize::MAX, None) @@ -501,7 +507,7 @@ mod tests { let mut bad = multi_insert_raw(1, 120, 16610, &[3]); bad.main_data[0] = 0; // strip CONTAINS_NEW_TUPLE: ImageOnly on fold b.stash_raw(1, bad).await.unwrap(); - inject_ordinary(&mut b, rfn, rel.clone()); + let stash = inject_ordinary(rfn, rel.clone()); let mut view = MapView { routes: HashMap::from_iter([(rel.rel_name.clone(), route())]), @@ -511,7 +517,10 @@ mod tests { let resolver = ToastResolver::disabled(); let path = tmp.path().join("1.plan"); let mut planner = Planner::create(path.clone(), 1 << 20, &mut view, &resolver).unwrap(); - let mut drain = b.drain_committed(1, 42, 0x2000, &[], false).await.unwrap(); + let mut drain = b + .drain_committed(stash, 42, 0x2000, &[], false) + .await + .unwrap(); let mut planned = 0usize; // 1-row slices: valid fanout rows plan before the bad record folds let err = loop { @@ -556,7 +565,10 @@ mod tests { let resolver = ToastResolver::disabled(); let mut planner = Planner::create(tmp.path().join("1.plan"), 1 << 30, &mut view, &resolver).unwrap(); - let mut drain = b.drain_committed(1, 42, 0x2000, &[], false).await.unwrap(); + let mut drain = b + .drain_committed(StashResolved::nothing_stashed(1), 42, 0x2000, &[], false) + .await + .unwrap(); while let Some(batch) = drain.next_batch(1024, usize::MAX, None).await.unwrap() { let is_final = batch.is_final; planner.plan_batch(batch).await.unwrap(); @@ -631,7 +643,10 @@ mod tests { let resolver = ToastResolver::disabled(); let mut planner = Planner::create(tmp.path().join("1.plan"), 1 << 20, &mut view, &resolver).unwrap(); - let mut drain = b.drain_committed(1, 42, 0x2000, &[], true).await.unwrap(); + let mut drain = b + .drain_committed(StashResolved::nothing_stashed(1), 42, 0x2000, &[], true) + .await + .unwrap(); while let Some(batch) = drain.next_batch(1, usize::MAX, None).await.unwrap() { let is_final = batch.is_final; planner.plan_batch(batch).await.unwrap(); @@ -682,7 +697,10 @@ mod tests { let resolver = ToastResolver::disabled(); let path = tmp.path().join("1.plan"); let mut planner = Planner::create(path.clone(), 1 << 20, &mut view, &resolver).unwrap(); - let mut drain = b.drain_committed(1, 42, 0x2000, &[], false).await.unwrap(); + let mut drain = b + .drain_committed(StashResolved::nothing_stashed(1), 42, 0x2000, &[], false) + .await + .unwrap(); let batch = drain .next_batch(8, usize::MAX, None) .await @@ -733,7 +751,10 @@ mod tests { let resolver = ToastResolver::disabled(); let mut planner = Planner::create(tmp.path().join("1.plan"), 1 << 20, &mut view, &resolver).unwrap(); - let mut drain = b.drain_committed(1, 42, 0x2000, &[], false).await.unwrap(); + let mut drain = b + .drain_committed(StashResolved::nothing_stashed(1), 42, 0x2000, &[], false) + .await + .unwrap(); while let Some(batch) = drain.next_batch(8, usize::MAX, None).await.unwrap() { planner.plan_batch(batch).await.unwrap(); } diff --git a/src/emit/pipeline/reorder.rs b/src/emit/pipeline/reorder.rs index 397a3490..82d0f540 100644 --- a/src/emit/pipeline/reorder.rs +++ b/src/emit/pipeline/reorder.rs @@ -25,8 +25,8 @@ use walrus::pg::walparser::RmId; use crate::backfill::backfill_staging::StagingSession; use crate::backfill::visibility_pending::{self, SharedPendingLedger}; use crate::catalog::pending::PendingCatalog; -use crate::decode::heap_decoder::{DescribedHeap, HeapOp}; -use crate::decode::visibility::{PgXactPatch, PgXactView, read_pg_xact}; +use crate::decode::heap_decoder::DescribedHeap; +use crate::decode::visibility::{PgXactView, SealedPatch, read_pg_xact}; use crate::emit::ch_ddl::DdlApplicator; use crate::emit::ch_emitter::EmitterStats; use crate::record::{Record, RecordSink, SinkError}; @@ -43,10 +43,10 @@ use crate::xact::xact_buffer::{DrainEntry, SubxactTracker, XactBuffer}; use crate::config::ResolvedConfig; use crate::emit::pipeline::Fatal; -use crate::emit::pipeline::ack::AckHandle; +use crate::emit::pipeline::ack::{AckHandle, Publish, SeqAlloc}; use crate::emit::pipeline::batcher::BatcherMsg; use crate::emit::pipeline::decode; -use crate::emit::pipeline::plan_spool::{PlanItem, SealedPlan}; +use crate::emit::pipeline::plan_spool::{PlanItem, SealedPlan, VerifiedPlan}; use crate::emit::pipeline::planner::{PlanRouteView, Planner, drain_reason}; use crate::emit::route::{RouteSnapshot, RoutedHeap, RowPolicy}; use crate::mapping::{MappingSnapshot, TableMapping}; @@ -112,8 +112,8 @@ pub struct ReorderSink { /// a crash-now restart resumes from. Seeded at the resolved start, /// advanced only after each manifest persist. resume_floor: Arc>, - /// Dense commit-order counter; one seq per dispatched data unit. - next_seq: u64, + /// One seq per dispatched data unit, in commit order + seqs: SeqAlloc, /// Drain-slice budget: rows / bytes per [`DrainedBatch`] pulled from the /// buffer. Bounds resident decoded rows while a spilled xact streams /// back. @@ -138,9 +138,21 @@ pub struct ReorderSink { route_config: Option>, } +/// Every earlier row is durable on CH, so a DDL, TRUNCATE or config apply +/// orders strictly after the data before it +struct Fenced(()); + +impl Fenced { + /// Boot runs before the pump dispatches any row + const fn pre_pump() -> Self { + Self(()) + } +} + impl ReorderSink { + /// Sink only usable after [`BootingReorder::boot`] #[allow(clippy::too_many_arguments)] - pub fn new( + pub(super) fn booting( buffer: Arc>, dbs: Arc, pending: Arc, @@ -164,7 +176,7 @@ impl ReorderSink { pending_rows: SharedPendingLedger, dest: Arc, resume_floor: Arc>, - ) -> Self { + ) -> BootingReorder { // subscribe() marks the current value seen, so a `ctl reload` // racing pipeline spawn would stay invisible to has_changed — // and seeding the applied set from that raced value would record @@ -192,7 +204,8 @@ impl ReorderSink { (db.oid, scope) }) .collect(); - Self { + let seqs = ack.seqs(0); + BootingReorder(Self { buffer, dbs, scopes, @@ -203,7 +216,7 @@ impl ReorderSink { stats, resolver, fatal, - next_seq: 0, + seqs, batch_rows, batch_bytes, plan_disk_max, @@ -217,7 +230,7 @@ impl ReorderSink { resume_floor, route_mapping: None, route_config: None, - } + }) } /// Route-point steps 1–2 close out here: preceding schema/config state is @@ -374,12 +387,6 @@ impl ReorderSink { Ok(()) } - fn alloc_seq(&mut self) -> u64 { - let s = self.next_seq; - self.next_seq += 1; - s - } - fn fatal_err(&self) -> SinkError { SinkError::Other( self.fatal @@ -400,6 +407,7 @@ impl ReorderSink { /// poison the sink future's `Send` bound. async fn apply_config( &mut self, + _: Fenced, db: Oid, event: &ConfigEvent, commit_lsn: u64, @@ -460,10 +468,10 @@ impl ReorderSink { } // Barrier waits prefer concurrent fatal over successful completion - /// Block until every seq `< self.next_seq` is durable on CH, or a fatal - /// trips (e.g. CH down past the inserter retry budget). + /// Block until every opened seq is durable on CH, or a fatal trips (e.g. + /// CH down past the inserter retry budget). async fn wait_all_durable(&mut self) -> Result<(), SinkError> { - let through = self.next_seq; + let through = self.seqs.next(); tokio::select! { biased; _ = self.fatal.wait() => Err(self.fatal_err()), @@ -476,13 +484,15 @@ impl ReorderSink { /// Fence before applying a DDL event / TRUNCATE so it orders strictly /// after all earlier data: seal batcher, wait durable. Rows place inline /// before dispatch returns, so `FlushAll` orders after all of them - async fn barrier_fence(&mut self) -> Result<(), SinkError> { + async fn barrier_fence(&mut self) -> Result { self.flush_all_batcher().await?; - self.wait_all_durable().await + self.wait_all_durable().await?; + Ok(Fenced(())) } async fn apply_event( &mut self, + _: Fenced, db: Oid, event: &SchemaEvent, commit_lsn: u64, @@ -546,10 +556,10 @@ impl ReorderSink { /// Boot Added pass: every relation `Present` in the descriptor log at /// resume gets an `Added` apply (idempotent `CREATE TABLE IF NOT /// EXISTS` + forward-declaration materialise). Runs pre-pump every - /// boot, like [`Self::flush_due_retires`] — brownfield auto-create + /// boot, like [`Self::flush_due_retires`], brownfield auto-create /// tables exist at attach instead of first write, and newly enabled /// auto-create/mapping picks up existing rels without log mutation. - pub async fn apply_boot_events( + async fn apply_boot_events( &mut self, descs: Vec>, resume_lsn: u64, @@ -559,8 +569,13 @@ impl ReorderSink { continue; } let db = desc.rfn.db_node; - self.apply_event(db, &SchemaEvent::Added { desc }, resume_lsn) - .await?; + self.apply_event( + Fenced::pre_pump(), + db, + &SchemaEvent::Added { desc }, + resume_lsn, + ) + .await?; } Ok(()) } @@ -568,11 +583,11 @@ impl ReorderSink { /// Retire below persisted resolved floor /// /// Restart resumes at the floor, DROP lock excludes later referrers. - /// Pub for boot: entries due at resume must retire during standup — + /// Boot runs it too: entries due at resume must retire during standup, /// their drop never replays, so no commit re-triggers this flush. /// Ledger removal persists after each wipe; a crash between re-runs /// an idempotent `TRUNCATE` on the emptied mirror - pub async fn flush_due_retires(&mut self) -> Result<(), SinkError> { + async fn flush_due_retires(&mut self) -> Result<(), SinkError> { // Disabled resolver no-ops retire_mirror: flushing would drop ledger // entries without wiping mirrors, leaking them for a later CH run // over the same spill dir @@ -620,21 +635,21 @@ impl ReorderSink { if ledger.is_empty() { return Ok(()); } - if self.pending_session.is_none() { - self.pending_session = Some( + let sess = match &mut self.pending_session { + Some(sess) => sess, + None => self.pending_session.insert( StagingSession::connect(self.dest.clone()) .await .map_err(|e| SinkError::Other(format!("pending visibility: connect: {e}")))?, - ); - } - let sess = self.pending_session.as_mut().expect("just connected"); + ), + }; visibility_pending::settle(&mut ledger, sess, &self.stats) .await .map_err(SinkError::Other) } /// Recover outcomes from shadow transaction logs before resumed WAL - pub async fn settle_pending_boot( + async fn settle_pending_boot( &mut self, shadow_data_dir: Option<&std::path::Path>, ) -> Result<(), SinkError> { @@ -645,7 +660,7 @@ impl ReorderSink { let accum = read_pg_xact(dir).await.map_err(|e| { SinkError::Other(format!("pending visibility: shadow pg_xact: {e}")) })?; - let patch = PgXactPatch::new(); + let patch = SealedPatch::default(); let fold = self .pending_rows .lock() @@ -684,13 +699,14 @@ impl ReorderSink { async fn apply_drain_entry( &mut self, + fenced: Fenced, db: Oid, entry: &DrainEntry, commit_lsn: u64, ) -> Result<(), SinkError> { match entry { - DrainEntry::Catalog(ev) => self.apply_event(db, ev, commit_lsn).await, - DrainEntry::Config(ev) => self.apply_config(db, ev, commit_lsn).await, + DrainEntry::Catalog(ev) => self.apply_event(fenced, db, ev, commit_lsn).await, + DrainEntry::Config(ev) => self.apply_config(fenced, db, ev, commit_lsn).await, DrainEntry::ToastBarrier { toast_relid, marker_lsn, @@ -701,7 +717,7 @@ impl ReorderSink { } } - async fn apply_truncate(&mut self, heap: &DescribedHeap) -> Result<(), SinkError> { + async fn apply_truncate(&mut self, _: Fenced, heap: &DescribedHeap) -> Result<(), SinkError> { let Some(scope) = self.scopes.get_mut(&heap.descriptor.rfn.db_node) else { return Ok(()); }; @@ -764,7 +780,7 @@ impl ReorderSink { pending_bytes: &mut usize, commit_ts: i64, commit_lsn: u64, - publish: bool, + publish: Publish, ) -> Result<(), SinkError> { if pending.is_empty() { return Ok(()); @@ -776,12 +792,7 @@ impl ReorderSink { _ = self.fatal.wait() => return Err(self.fatal_err()), p = crate::budget::admit_opt(self.budget.as_ref(), bytes) => p.map(Arc::new), }; - let seq = self.alloc_seq(); - if publish { - self.ack.register(seq, commit_lsn); - } else { - self.ack.register_partial(seq, commit_lsn); - } + let seq = self.seqs.open(commit_lsn, publish); self.stats.queue_jobs_out.fetch_add(1, Ordering::Relaxed); let chunk_rows = self.dest.current().decode_chunk_rows; let rows = tokio::select! { @@ -791,14 +802,14 @@ impl ReorderSink { &self.msg_tx, &self.stats, chunk_rows, - seq, + seq.seq(), commit_ts, commit_lsn, heaps, permit, ) => r.map_err(SinkError::Other)?, }; - self.ack.placed(seq, rows); + seq.place(rows); Ok(()) } @@ -812,23 +823,15 @@ impl ReorderSink { pub async fn execute_plan( &mut self, db: Oid, - plan: &SealedPlan, + plan: &VerifiedPlan, ) -> Result<(u64, bool), SinkError> { let (commit_ts, commit_lsn) = (plan.commit_ts, plan.commit_lsn); - // Mem-resident plans hold the bytes validated at write; file-backed - // plans re-read from disk, checksum-verify fully before the first - // side effect so corruption fails the whole transaction - if plan.path().is_some() { - plan.verify() - .map_err(|e| SinkError::Other(format!("plan verify: {e}")))?; - } let mut rd = plan .replay() .map_err(|e| SinkError::Other(format!("plan replay: {e}")))?; let mut pending: Vec = Vec::new(); let mut pending_bytes = 0usize; let mut rows_cursor = 0usize; - let mut trunc = plan.truncate_rows.iter().copied(); let total_rows: usize = plan.row_batches.iter().map(|rb| rb.len()).sum(); let mut rows_total = 0u64; while let Some(item) = rd @@ -844,25 +847,25 @@ impl ReorderSink { &mut pending_bytes, commit_ts, commit_lsn, - false, + Publish::Partial, ) .await?; - self.barrier_fence().await?; - self.apply_drain_entry(db, &c.event, commit_lsn).await?; + let fenced = self.barrier_fence().await?; + self.apply_drain_entry(fenced, db, &c.event, commit_lsn) + .await?; } - PlanItem::Heap(h) if matches!(h.described.decoded.op, HeapOp::Truncate) => { - let upto = trunc.next().unwrap_or(rows_cursor); + PlanItem::Truncate { heap: h, upto } => { self.put_plan_rows(plan, &mut rows_cursor, upto).await?; self.dispatch_planned( &mut pending, &mut pending_bytes, commit_ts, commit_lsn, - false, + Publish::Partial, ) .await?; - self.barrier_fence().await?; - self.apply_truncate(&h.described).await?; + let fenced = self.barrier_fence().await?; + self.apply_truncate(fenced, &h.described).await?; } PlanItem::Heap(h) => { rows_total += 1; @@ -874,7 +877,7 @@ impl ReorderSink { &mut pending_bytes, commit_ts, commit_lsn, - false, + Publish::Partial, ) .await?; } @@ -883,16 +886,16 @@ impl ReorderSink { } self.put_plan_rows(plan, &mut rows_cursor, total_rows) .await?; - let publish = !pending.is_empty(); + let published = !pending.is_empty(); self.dispatch_planned( &mut pending, &mut pending_bytes, commit_ts, commit_lsn, - publish, + Publish::Commit, ) .await?; - Ok((rows_total, publish)) + Ok((rows_total, published)) } async fn on_commit( @@ -925,7 +928,7 @@ impl ReorderSink { .get(&db) .map(|scope| scope.db.desc_log.clone()) .unwrap_or_else(|| self.dbs.primary().desc_log.clone()); - crate::xact::xact_buffer::resolve_stash( + let stash = crate::xact::xact_buffer::resolve_stash( &self.buffer, &stash_log, &self.pending, @@ -954,7 +957,7 @@ impl ReorderSink { let mut drain = { let mut buf = self.buffer.lock().await; buf.drain_committed( - xid, + stash, payload.xact_time, record.source_lsn, &payload.subxacts, @@ -994,7 +997,7 @@ impl ReorderSink { self.reset_route_state(db).await; let mut rows_total: u64 = 0; let mut published = false; - if drain.had_states { + if drain.had_states() { let plan_path = self.plan_dir.join(format!("xact-{xid}-{commit_lsn}.plan")); let plan = { let scope = self.scopes.get_mut(&db); @@ -1055,6 +1058,10 @@ impl ReorderSink { &self.stats.plan_bytes_mem }; plan_bytes.fetch_add(plan.size_bytes, Ordering::Relaxed); + // Corruption fails the whole transaction before any side effect + let plan = plan + .into_verified() + .map_err(|e| SinkError::Other(format!("plan verify: {e}")))?; (rows_total, published) = self .execute_plan(db, &plan) .instrument(trace_span!( @@ -1073,9 +1080,7 @@ impl ReorderSink { // rows=0 marker: publishes commit_lsn once every earlier partial // segment is durable. Covers empty / read-only commits and plans // whose tail is a control or truncate. - let seq = self.alloc_seq(); - self.ack.register(seq, commit_lsn); - self.ack.placed(seq, 0); + self.seqs.mark(commit_lsn); } Ok(()) } @@ -1088,21 +1093,51 @@ impl ReorderSink { // ABORT PREPARED: buffered state keys off the prepared xid let xid = payload.twophase_xid.unwrap_or(xid); self.note_pending(xid, &payload.subxacts, false).await?; - let seq = self.alloc_seq(); - self.ack.register(seq, record.source_lsn); + let seq = self.seqs.open(record.source_lsn, Publish::Commit); { let mut buf = self.buffer.lock().await; buf.abort(xid, Pos::new(record.source_lsn), &payload.subxacts) .await .map_err(SinkError::from)?; } - self.ack.placed(seq, 0); + seq.place(0); self.subxact_tracker.lock().await.forget_tree(xid); self.pending.forget_tree(xid); Ok(()) } } +/// Reorder sink awaiting its boot pass; only [`Self::boot`] yields the +/// [`RecordSink`] +pub struct BootingReorder(ReorderSink); + +impl BootingReorder { + /// Retire due mirrors, recover pending row outcomes, then apply the + /// Added pass over relations `present` at `resume_lsn` + pub async fn boot( + self, + shadow_data_dir: Option<&std::path::Path>, + present: Vec>, + resume_lsn: u64, + ) -> Result { + let mut sink = self.0; + sink.flush_due_retires() + .await + .map_err(|e| boot_step("flush of due toast-mirror retires", e))?; + sink.settle_pending_boot(shadow_data_dir) + .await + .map_err(|e| boot_step("settle of pending backup rows", e))?; + sink.apply_boot_events(present, resume_lsn) + .await + .map_err(|e| boot_step("Added pass over descriptor log", e))?; + Ok(sink) + } +} + +fn boot_step(step: &str, e: SinkError) -> SinkError { + SinkError::Other(format!("boot {step}: {e}")) +} + /// Route a plan-failure reason label onto its counter fn bump_plan_failure(stats: &EmitterStats, reason: &'static str) { let counter = match reason { @@ -1277,7 +1312,7 @@ impl RecordSink for ReorderSink { #[cfg(test)] mod tests { use super::*; - use crate::decode::heap_decoder::DecodedHeap; + use crate::decode::heap_decoder::{DecodedHeap, HeapOp}; use crate::mapping::TableTarget; use crate::xact::xact_buffer::raw_fixtures::int4_descriptor; diff --git a/src/emit/pipeline/tail.rs b/src/emit/pipeline/tail.rs index f22bc0be..6faeb131 100644 --- a/src/emit/pipeline/tail.rs +++ b/src/emit/pipeline/tail.rs @@ -38,10 +38,16 @@ pub struct TailParts { } impl TailParts { - /// Await the drain cascade. Call only after every producer-held `msg_tx` - /// and `AckHandle` clone has dropped, else the batcher never sees its - /// channel close and this hangs. - pub async fn join(self) { + /// Drop the spawn-returned producer handles, then await the drain + /// cascade. Hangs while any clone handed to a producer stays alive + pub async fn close(self, msg_tx: mpsc::Sender, ack: AckHandle) { + drop(msg_tx); + drop(ack); + self.join().await; + } + + /// Await the drain cascade once every producer handle has dropped + pub(super) async fn join(self) { let _ = self.batcher.await; for h in self.resolvers { let _ = h.await; @@ -64,9 +70,7 @@ impl TailParts { fatal: &Fatal, ) -> Result<(), String> { flush_and_prove(&msg_tx, &ack, through, fatal).await?; - drop(msg_tx); - drop(ack); - self.join().await; + self.close(msg_tx, ack).await; if let Some(msg) = fatal.message() { return Err(msg); } @@ -181,9 +185,7 @@ impl OwnedTail { /// down. Bounded by the inserters' retry policy (a CH outage trips their /// fatal, not a hang) pub async fn quiesce(self) { - drop(self.msg_tx); - drop(self.ack); - self.parts.join().await; + self.parts.close(self.msg_tx, self.ack).await; } } @@ -227,13 +229,13 @@ pub fn spawn_null( let batcher = tokio::spawn(async move { while let Some(msg) = msg_rx.recv().await { match msg { - BatcherMsg::Row(r) => swallow_ack.acked(vec![(r.seq, 1)]), + BatcherMsg::Row(r) => swallow_ack.swallowed(vec![(r.seq, 1)]), BatcherMsg::Rows(chunk) => { let mut counts: HashMap = HashMap::new(); for r in &chunk.rows { *counts.entry(r.seq).or_insert(0) += 1; } - swallow_ack.acked(counts.into_iter().collect()); + swallow_ack.swallowed(counts.into_iter().collect()); } BatcherMsg::FlushAll(reply) => { let _ = reply.send(()); diff --git a/src/filter/engine.rs b/src/filter/engine.rs index a6bb6e84..c177b7f1 100644 --- a/src/filter/engine.rs +++ b/src/filter/engine.rs @@ -71,12 +71,10 @@ impl FilterStats { #[derive(Debug, Clone)] pub struct Verdict { pub route: Route, - /// Commit record of a catalog-mutating xact (top, subxact, or prepared - /// xid wrote a catalog-touching record), or a command boundary inside - /// one. Pump holds shadow publication here - /// until replay passes the record's `next_lsn` - pub catalog_boundary: bool, - /// Capture input; `Some` iff `catalog_boundary` + /// Capture input at a commit record of a catalog-mutating xact (top, + /// subxact, or prepared xid wrote a catalog-touching record), or a command + /// boundary inside one. Pump holds shadow publication here until replay + /// passes the record's `next_lsn` pub boundary: Option>, /// Members of a catalog-dirty tree this record aborted pub aborted_tree: Option>>, @@ -375,7 +373,6 @@ impl Filter { .record(class, route, record.header.total_record_length as u64); Ok(Verdict { route, - catalog_boundary: end.boundary.is_some(), boundary: end.boundary, aborted_tree: end.aborted_tree, defer_catalog_decode, @@ -1099,16 +1096,16 @@ mod tests { let v = f .decide_record(&xact_end(XLOG_XACT_COMMIT, 8, &[], None), 0, 0xD116) .unwrap(); - assert!(!v.catalog_boundary); + assert!(!v.boundary.is_some()); // Catalog-dirty xid 7 commit: boundary, drained after let v = f .decide_record(&xact_end(XLOG_XACT_COMMIT, 7, &[], None), 0, 0xD116) .unwrap(); - assert!(v.catalog_boundary); + assert!(v.boundary.is_some()); let v = f .decide_record(&xact_end(XLOG_XACT_COMMIT, 7, &[], None), 0, 0xD116) .unwrap(); - assert!(!v.catalog_boundary, "dirty mark consumed once"); + assert!(!v.boundary.is_some(), "dirty mark consumed once"); } #[test] @@ -1119,11 +1116,11 @@ mod tests { let v = f .decide_record(&xact_end(XLOG_XACT_ABORT, 7, &[], None), 0, 0xD116) .unwrap(); - assert!(!v.catalog_boundary, "rolled-back DDL never holds"); + assert!(!v.boundary.is_some(), "rolled-back DDL never holds"); let v = f .decide_record(&xact_end(XLOG_XACT_COMMIT, 7, &[], None), 0, 0xD116) .unwrap(); - assert!(!v.catalog_boundary, "abort drained the mark"); + assert!(!v.boundary.is_some(), "abort drained the mark"); } #[test] @@ -1139,7 +1136,7 @@ mod tests { 0xD116, ) .unwrap(); - assert!(v.catalog_boundary); + assert!(v.boundary.is_some()); } #[test] @@ -1179,7 +1176,7 @@ mod tests { let v = f .decide_record(&xact_end(XLOG_XACT_COMMIT, 100, &[], None), 200, 0xD116) .unwrap(); - assert!(!v.catalog_boundary, "aborted child never bounds"); + assert!(!v.boundary.is_some(), "aborted child never bounds"); } #[test] @@ -1206,7 +1203,7 @@ mod tests { let v = f .decide_record(&xact_end(XLOG_XACT_COMMIT, 7, &[], None), 200, 0xD116) .unwrap(); - assert!(!v.catalog_boundary, "prune-only commit never bounds"); + assert!(!v.boundary.is_some(), "prune-only commit never bounds"); // Tuple lock and vacuum inplace stats: same treatment for info in [XLOG_HEAP_LOCK, XLOG_HEAP_INPLACE] { let mut r = rec_with_xid(RmId::Heap, &[(5, 1259)], 8); @@ -1257,7 +1254,7 @@ mod tests { 0xD116, ) .unwrap(); - assert!(v.catalog_boundary); + assert!(v.boundary.is_some()); assert!(!user(&mut f, 100), "commit clears the tree"); } @@ -1283,7 +1280,7 @@ mod tests { 0xD116, ) .unwrap(); - assert!(v.catalog_boundary); + assert!(v.boundary.is_some()); } #[test] @@ -1297,7 +1294,7 @@ mod tests { .decide_record(&xact_end(XLOG_XACT_COMMIT, 9, &[], None), 0, 0xD116) .unwrap(); assert!( - v.catalog_boundary, + v.boundary.is_some(), "VACUUM FULL relmap write holds at commit" ); } @@ -1315,7 +1312,7 @@ mod tests { .decide_record(&xact_end(XLOG_XACT_COMMIT, 7, &[], None), 0, 0xD116) .unwrap(); assert!( - !v.catalog_boundary, + !v.boundary.is_some(), "safe-default keep is not a catalog touch" ); } @@ -1329,7 +1326,7 @@ mod tests { let v = f .decide_record(&xact_end(XLOG_XACT_COMMIT, 7, &[], None), 0, 0xD116) .unwrap(); - assert!(v.catalog_boundary); + assert!(v.boundary.is_some()); } #[test] @@ -1712,11 +1709,11 @@ mod tests { let v = f .decide_record(&xact_end(XLOG_XACT_ABORT, 7, &[], None), 200, 0xD116) .unwrap(); - assert!(!v.catalog_boundary); + assert!(!v.boundary.is_some()); let v = f .decide_record(&xact_end(XLOG_XACT_COMMIT, 7, &[], None), 300, 0xD116) .unwrap(); - assert!(!v.catalog_boundary, "abort drained the mark"); + assert!(!v.boundary.is_some(), "abort drained the mark"); } #[test] @@ -1929,7 +1926,7 @@ mod tests { 0xD116, ) .unwrap(); - assert!(!v.catalog_boundary, "foreign commit bounds nothing local"); + assert!(!v.boundary.is_some(), "foreign commit bounds nothing local"); // Same oid in the followed database: dirt, fence, boundary, and a // first touch that owes nothing to the foreign write @@ -1994,7 +1991,7 @@ mod tests { let v = f .decide_record(&xact_end_dbinfo(XLOG_XACT_COMMIT, xid, db), 200, 0xD116) .unwrap(); - assert!(!v.catalog_boundary, "db {db} relmap is not target dirt"); + assert!(!v.boundary.is_some(), "db {db} relmap is not target dirt"); } } @@ -2023,7 +2020,7 @@ mod tests { 0xD116, ) .unwrap(); - assert!(!v.catalog_boundary, "commit drained the tree"); + assert!(!v.boundary.is_some(), "commit drained the tree"); } #[test] @@ -2046,7 +2043,7 @@ mod tests { let v = f .decide_record(&xact_end(XLOG_XACT_COMMIT, 7, &[], None), 300, 0xD116) .unwrap(); - assert!(!v.catalog_boundary, "drain ran before the scope check"); + assert!(!v.boundary.is_some(), "drain ran before the scope check"); } #[test] diff --git a/src/ops/bridge.rs b/src/ops/bridge.rs index 0301f66f..25ad8ea7 100644 --- a/src/ops/bridge.rs +++ b/src/ops/bridge.rs @@ -10,8 +10,8 @@ use std::io; use std::path::{Path, PathBuf}; +use std::sync::Arc; use std::sync::atomic::{AtomicU64, AtomicUsize, Ordering}; -use std::sync::{Arc, OnceLock}; use std::time::{Duration, Instant}; use backon::{ExponentialBuilder, Retryable}; @@ -325,10 +325,10 @@ pub struct Bridge { /// Stateless ops round-robin; [`Op::pinned`] ops stay on slot 0 slots: Vec, next: AtomicUsize, - /// Set by the first successful `HELLO`; later dials, on any slot, must - /// match it. A worker that came back a different build means a mixed - /// install and must fail closed - info: OnceLock, + /// Slot 0's `HELLO`; later dials, on any slot, must match it. A worker + /// that came back a different build means a mixed install and must fail + /// closed + info: Hello, bufs: BufPool, pub stats: Arc, } @@ -367,16 +367,17 @@ impl Bridge { conn: Mutex::new(None), }) .collect(); + let stats = Arc::new(BridgeStats::default()); + let (first, info) = Self::hello(&stats, &slots[0]).await?; + *slots[0].conn.lock().await = Some(first); let bridge = Self { bufs: BufPool::new(slots.len()), slots, next: AtomicUsize::new(0), - info: OnceLock::new(), - stats: Arc::new(BridgeStats::default()), + info, + stats, }; - // Slot 0 first, so its HELLO is the identity the rest are checked - // against rather than whichever worker happened to answer first - for slot in &bridge.slots { + for slot in &bridge.slots[1..] { let stream = bridge.dial(slot).await?; *slot.conn.lock().await = Some(stream); } @@ -396,10 +397,8 @@ impl Bridge { self.slots.len() } - /// `None` before the first successful `HELLO`, which [`connect`](Self::connect) - /// guarantees - pub fn info(&self) -> Option { - self.info.get().copied() + pub fn info(&self) -> Hello { + self.info } pub fn is_up(&self) -> bool { @@ -612,6 +611,19 @@ impl Bridge { /// Fresh socket plus `HELLO`. Takes no connection lock, so /// [`call`](Self::call) may hold one across it async fn dial(&self, slot: &Slot) -> Result { + let (stream, info) = Self::hello(&self.stats, slot).await?; + // A worker that came back a different build must not be trusted to + // answer requests the daemon framed against the old one + if info != self.info { + return Err(BridgeError::Protocol(format!( + "worker identity changed across reconnect: {:?} then {info:?}", + self.info + ))); + } + Ok(stream) + } + + async fn hello(stats: &BridgeStats, slot: &Slot) -> Result<(UnixStream, Hello), BridgeError> { let mut stream = UnixStream::connect(&slot.path).await?; widen_sockbufs(&stream); let started = Instant::now(); @@ -619,7 +631,7 @@ impl Bridge { patch_frame(&mut hello, Op::Hello); let mut body = Vec::new(); let res = round_trip(&mut stream, &hello, &mut body).await; - self.record(Op::Hello, started, &res); + record(stats, Op::Hello, started, &res); let len = res?; let mut c = Cursor::at(&body[..len], 1); @@ -640,15 +652,7 @@ impl Bridge { in_recovery: c.u8()? != 0, datid: c.u32()?, }; - // A worker that came back a different build must not be trusted to - // answer requests the daemon framed against the old one - let first = *self.info.get_or_init(|| info); - if first != info { - return Err(BridgeError::Protocol(format!( - "worker identity changed across reconnect: {first:?} then {info:?}" - ))); - } - Ok(stream) + Ok((stream, info)) } /// `frame` carries [`FRAME_PREFIX_BYTES`] of unwritten leading room, @@ -731,20 +735,23 @@ impl Bridge { } fn record(&self, op: Op, started: Instant, res: &Result) { - let slot = op.slot(); - self.stats.requests[slot].fetch_add(1, Ordering::Relaxed); - self.stats.request_nanos[slot] - .fetch_add(started.elapsed().as_nanos() as u64, Ordering::Relaxed); - if res.is_err() { - self.stats.errors[slot].fetch_add(1, Ordering::Relaxed); - } - // A worker that answered with an error status is still up. A frame - // this side refused never reached it, so it says nothing either way - if !matches!(res, Err(BridgeError::RequestTooLarge { .. })) { - self.stats - .up - .store(u64::from(!is_transport_error(res)), Ordering::Relaxed); - } + record(&self.stats, op, started, res); + } +} + +fn record(stats: &BridgeStats, op: Op, started: Instant, res: &Result) { + let slot = op.slot(); + stats.requests[slot].fetch_add(1, Ordering::Relaxed); + stats.request_nanos[slot].fetch_add(started.elapsed().as_nanos() as u64, Ordering::Relaxed); + if res.is_err() { + stats.errors[slot].fetch_add(1, Ordering::Relaxed); + } + // A worker that answered with an error status is still up. A frame + // this side refused never reached it, so it says nothing either way + if !matches!(res, Err(BridgeError::RequestTooLarge { .. })) { + stats + .up + .store(u64::from(!is_transport_error(res)), Ordering::Relaxed); } } diff --git a/src/ops/control.rs b/src/ops/control.rs index 3190d31d..b1c63f7b 100644 --- a/src/ops/control.rs +++ b/src/ops/control.rs @@ -772,6 +772,7 @@ mod tests { Some(path), Table::new(), crate::mapping::mapping_handle(Default::default()), + crate::config::ResolverBoot::default(), ); let reloader = Reloader::default(); reloader.set_resolvers(vec![resolver]).await; diff --git a/src/pos.rs b/src/pos.rs index 10f647bd..dc826a3d 100644 --- a/src/pos.rs +++ b/src/pos.rs @@ -64,6 +64,9 @@ lsn_kinds! { Snapshot => "snapshot"; /// Segment-aligned, archive-clamped resume and GC floor Floor => "floor"; + /// Unaligned resume pick: operator pin, bootstrap end, durable ack or + /// greenfield head, before alignment makes it a [`Floor`] + RawStart => "raw_start"; /// Timeline start or ancestor switchpoint Switchpoint => "switchpoint"; } @@ -71,12 +74,32 @@ lsn_kinds! { kinds! { /// Contiguous done prefix in ack collector seq space AckFrontier => "ack_frontier"; +} + +macro_rules! count_kinds { + ($($(#[$m:meta])* $name:ident => $label:literal;)+) => { + kinds! { $($(#[$m])* $name => $label;)+ } + $(impl CountKind for $name {})+ + }; +} + +count_kinds! { /// Enqueues onto shadow send queues QueuedWake => "queued_wake"; /// Standby status reports plus connection attaches AppliedWake => "applied_wake"; } +macro_rules! observed { + ($($name:ident),+ $(,)?) => { $(impl ObservedKind for $name {})+ }; +} + +observed! { + SourceReceived, ShadowReplay, ShadowFlush, ShadowDispatched, Drain, EmitterAck, + ResumeSafe, XactFirst, FilterDispatched, Commit, Snapshot, Switchpoint, + AckFrontier, QueuedWake, AppliedWake, +} + pub struct Pos(u64, PhantomData K>); impl Pos { @@ -175,6 +198,38 @@ impl<'de, K: LsnKind> Deserialize<'de> for Pos { } } +/// Kinds whose cells advance from any observed value. Durability watermarks +/// stay off it, their cells advance only through [`Durable`] +pub trait ObservedKind: PosKind {} + +/// Event tally, the only kind [`Monotone::bump`] accepts, so a byte position +/// can never step forward without a value backing it +pub trait CountKind: ObservedKind {} + +/// Position proven persisted. Constructed only where persistence happens, so +/// a cell publishing it can never run ahead of what a crash-now restart keeps +pub struct Durable(Pos); + +impl Durable { + pub(crate) const fn new(pos: Pos) -> Self { + Self(pos) + } + + /// Integration tests stand in for persistence they do not run + #[cfg(feature = "test-support")] + pub const fn assume_for_test(pos: Pos) -> Self { + Self(pos) + } +} + +impl Clone for Durable { + fn clone(&self) -> Self { + *self + } +} + +impl Copy for Durable {} + /// Shared position that only moves forward, with level-triggered waiters /// /// `join` is the routine mutator: a watermark that can be assigned is one that @@ -193,7 +248,7 @@ impl Monotone { } } - pub fn join(&self, v: Pos) -> Pos { + fn advance_to(&self, v: Pos) -> Pos { let v = v.get(); let mut prev = 0; self.tx.send_if_modified(|cur| { @@ -203,15 +258,14 @@ impl Monotone { Pos::new(prev.max(v)) } - /// Re-anchor after position space changes, such as timeline fork - pub fn rebase(&self, v: Pos) -> Pos { - self.tx.send_replace(v.get()); - v + pub fn publish(&self, v: Durable) -> Pos { + self.advance_to(v.0) } - /// Advance a cell that counts events rather than bytes - pub fn bump(&self) { - self.tx.send_modify(|cur| *cur += 1); + /// Re-anchor after position space changes, such as timeline fork + pub fn rebase(&self, v: Durable) -> Pos { + self.tx.send_replace(v.0.get()); + v.0 } pub fn get(&self) -> Pos { @@ -226,6 +280,18 @@ impl Monotone { } } +impl Monotone { + pub fn join(&self, v: Pos) -> Pos { + self.advance_to(v) + } +} + +impl Monotone { + pub fn bump(&self) { + self.tx.send_modify(|cur| *cur += 1); + } +} + impl Default for Monotone { fn default() -> Self { Self::new(Pos::ZERO) @@ -253,12 +319,6 @@ impl Clone for Gate { } } -impl fmt::Debug for Gate { - fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { - self.current().fmt(f) - } -} - /// Cell dropped before the target was covered, so nothing can reach it #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub struct GateClosed; @@ -305,9 +365,13 @@ mod tests { #[test] fn only_rebase_lowers_a_cell() { let m: Monotone = Monotone::default(); - assert_eq!(m.join(Pos::new(500u64)), 500); - assert_eq!(m.join(Pos::new(100u64)), 500); - assert_eq!(m.rebase(Pos::new(100u64)), 100, "fork re-anchors the floor"); + assert_eq!(m.publish(Durable::new(Pos::new(500u64))), 500); + assert_eq!(m.publish(Durable::new(Pos::new(100u64))), 500); + assert_eq!( + m.rebase(Durable::new(Pos::new(100u64))), + 100, + "fork re-anchors the floor" + ); assert_eq!(m.get(), 100); } diff --git a/src/record.rs b/src/record.rs index 1f468728..9cacf598 100644 --- a/src/record.rs +++ b/src/record.rs @@ -180,10 +180,9 @@ pub struct Record<'a> { pub next_lsn: u64, pub page_magic: u16, pub route: Route, - /// Commit changed catalog data, so pause publication of later bytes - /// until shadow replays through `next_lsn`, except for statistics-only commits - pub catalog_boundary: bool, - /// Capture input for a catalog boundary; `Some` iff `catalog_boundary` + /// Catalog boundary: commit or command changed catalog data, so pause + /// publication of later bytes until shadow replays through `next_lsn`, + /// except for statistics-only commits pub boundary_info: Option>, /// Abort of a catalog-dirty tree: every member the filter drained. /// Pending catalog slots those xids wrote drop here, on the pump, ahead @@ -262,7 +261,6 @@ impl RecordSink for CollectingRecordSink { next_lsn: record.next_lsn, page_magic: record.page_magic, route: record.route, - catalog_boundary: record.catalog_boundary, boundary_info: record.boundary_info.clone(), aborted_tree: record.aborted_tree.clone(), defer_catalog_decode: record.defer_catalog_decode, diff --git a/src/runtime_config.rs b/src/runtime_config.rs index 23100af5..6e572080 100644 --- a/src/runtime_config.rs +++ b/src/runtime_config.rs @@ -4,7 +4,7 @@ //! //! Config rows live in operator-owned `.config_*` tables on source PG //! (see `sql/runtime_config_install.sql`). The daemon reads them at boot -//! (`SELECT *`, [`crate::config::ConfigResolver::seed_overlay`]) and tracks +//! (`SELECT *`, [`crate::config::ResolverBoot::overlay`]) and tracks //! live edits off the WAL stream: a config-table heap write is detected in the //! decode path by resolved qualified name, interpreted here into a //! [`ConfigEvent`], and applied at the row's commit LSN. diff --git a/src/source/archive.rs b/src/source/archive.rs index f1bc5f4c..8e14bb51 100644 --- a/src/source/archive.rs +++ b/src/source/archive.rs @@ -180,12 +180,16 @@ fn extends(known: &TimelineHistory, next: &TimelineHistory) -> bool { } /// Check that archive and source have matching timeline histories, since replay -/// uses source history to choose archived segment names. +/// uses source history to choose archived segment names. PG writes no history +/// file for timeline 1, so a never-promoted source has nothing to compare pub async fn verify_history( settings: &Settings, storage: &DynStorage, source: &TimelineHistory, ) -> Result<()> { + if source.target() == 1 { + return Ok(()); + } let name = history_filename(source.target()); let raw = walrus::pg::wal::fetch::read_segment(settings, storage, &name) .await @@ -278,6 +282,15 @@ mod tests { assert!(err.to_string().contains("00000003.history"), "{err:#}"); } + #[tokio::test] + async fn timeline_one_needs_no_archived_history() { + let tmp = tempfile::tempdir().unwrap(); + let (settings, storage) = fs_archive(&tmp.path().join("archive")); + verify_history(&settings, &storage, &TimelineHistory::root(1)) + .await + .unwrap(); + } + #[tokio::test] async fn archived_ids_need_not_be_consecutive() { let tmp = tempfile::tempdir().unwrap(); diff --git a/src/source/boundary_hold.rs b/src/source/boundary_hold.rs index ee292f12..58bc50d5 100644 --- a/src/source/boundary_hold.rs +++ b/src/source/boundary_hold.rs @@ -1,6 +1,6 @@ //! Catalog-boundary publication hold. //! -//! At the commit of a catalog-mutating xact ([`Record::catalog_boundary`], +//! At the commit of a catalog-mutating xact ([`Record::boundary_info`], //! stamped by the pump classifier) the pump must not publish successor //! bytes until shadow replays through the commit's `next_lsn` //! (PG `EndRecPtr`). [`BoundaryHoldSink`] enforces this by blocking inside @@ -32,6 +32,7 @@ use std::time::{Duration, Instant}; use tokio::sync::Mutex; +use crate::catalog::shadow_catalog::ParkedAt; use crate::pos::{Pos, ShadowReplay}; use crate::record::{BoundaryKind, Record, RecordSink, SinkError}; use crate::source::catalog_capture::CaptureSet; @@ -105,7 +106,7 @@ impl CatalogBoundaryGate { commit_lsn: u64, next_lsn: Pos, worker_alive: impl Fn() -> bool, - ) -> Result<(), SinkError> { + ) -> Result { let applied = self.state.lock().await.applied(); let start = Instant::now(); let held = loop { @@ -145,10 +146,10 @@ impl CatalogBoundaryGate { held = ?held, "catalog boundary released", ); - Ok(()) + Ok(ParkedAt::new(next_lsn.get())) } - fn fail(&self, msg: String) -> Result<(), SinkError> { + fn fail(&self, msg: String) -> Result { self.stats.failures.fetch_add(1, Ordering::Relaxed); Err(SinkError::Other(msg)) } @@ -219,26 +220,21 @@ impl RecordSink for BoundaryHoldSink { if let Some(members) = &record.aborted_tree { self.capture.forget_aborted(members); } - if !record.catalog_boundary { + let Some(info) = &record.boundary_info else { return self.inner.on_record(record).await; - } - let boundary = record.boundary_info.clone(); - let park = match &boundary { - Some(info) if !self.capture.is_empty() => { - self.capture.admits(info, record.next_lsn) - } + }; + let park = if self.capture.is_empty() { // Without capture, wait only for commits that change more than statistics - Some(info) => matches!(info.kind, BoundaryKind::Commit) && !info.stats_only, - None => true, + matches!(info.kind, BoundaryKind::Commit) && !info.stats_only + } else { + self.capture.admits(info, record.next_lsn) }; if !park { // Save an empty batch so restart can replay this commit // without reading catalog state from shadow - if let Some(info) = &boundary - && matches!(info.kind, BoundaryKind::Commit) - { + if matches!(info.kind, BoundaryKind::Commit) { self.capture - .capture_boundary(info, record.source_lsn, record.next_lsn) + .cover_unparked(info, record.source_lsn, record.next_lsn) .await?; } return self.inner.on_record(record).await; @@ -249,24 +245,25 @@ impl RecordSink for BoundaryHoldSink { self.inner.flush().await?; let inner = &self.inner; let parked = Instant::now(); - if let Err(hold_err) = self + let held = self .gate .hold(record.source_lsn, Pos::new(record.next_lsn), || { inner.worker_alive() }) - .await - { - // Prefer the worker's parked root cause over the generic - // hold error (empty-buffer flush only drains the err slot) - self.inner.flush().await?; - return Err(hold_err); - } - if let Some(info) = &boundary { - self.capture.charge_hold(info, parked.elapsed()); - self.capture - .capture_boundary(info, record.source_lsn, record.next_lsn) - .await?; - } + .await; + let held = match held { + Ok(held) => held, + Err(hold_err) => { + // Prefer the worker's parked root cause over the generic + // hold error (empty-buffer flush only drains the err slot) + self.inner.flush().await?; + return Err(hold_err); + } + }; + self.capture.charge_hold(info, parked.elapsed()); + self.capture + .capture_boundary(info, record.source_lsn, &held) + .await?; self.inner.on_record(record).await?; // Ship the commit immediately: the drain (and its CH effects) // must not wait out the accumulator @@ -465,7 +462,7 @@ mod tests { #[tokio::test] async fn sink_parks_only_at_catalog_boundary() { - // DML-only records (catalog_boundary = false) pass straight + // DML-only records (no boundary_info) pass straight // through with no shadow connection and no timeout let s = state(); let q = QueueingRecordSink::spawn(CountingRecordSink::default(), 4, 16, None); @@ -493,7 +490,6 @@ mod tests { let rec = Record { source_lsn: 0x1F00, next_lsn: 0x2000, - catalog_boundary: true, boundary_info: Some(Arc::new(crate::record::BoundaryInfo { kind: BoundaryKind::Command { writer_xid: 7 }, ..Default::default() @@ -519,7 +515,6 @@ mod tests { let rec = Record { source_lsn: 0x1F00, next_lsn: 0x2000, - catalog_boundary: true, boundary_info: Some(Arc::new(stub)), ..Default::default() }; @@ -568,7 +563,7 @@ mod tests { let rec = Record { source_lsn: 0x1F00, next_lsn: 0x2000, - catalog_boundary: true, + boundary_info: Some(Arc::new(crate::record::BoundaryInfo::default())), ..Default::default() }; sink.on_record(&rec).await.expect("boundary releases"); @@ -612,7 +607,7 @@ mod tests { let rec = Record { source_lsn: 0x1F00, next_lsn: 0x2000, - catalog_boundary: true, + boundary_info: Some(Arc::new(crate::record::BoundaryInfo::default())), ..Default::default() }; let err = sink.on_record(&rec).await.expect_err("must fail"); diff --git a/src/source/catalog_capture.rs b/src/source/catalog_capture.rs index c711f040..694c0e1e 100644 --- a/src/source/catalog_capture.rs +++ b/src/source/catalog_capture.rs @@ -58,7 +58,7 @@ use crate::catalog::desc_log::{ ObservationKind, RelationObservation, }; use crate::catalog::pending::{DegradeReason, PendingCatalog, PendingSlot}; -use crate::catalog::shadow_catalog::{CatalogError, ShadowCatalog}; +use crate::catalog::shadow_catalog::{CatalogError, ParkedAt, ShadowCatalog}; use crate::filter::SmgrMarkers; use crate::ops::bridge::BridgeError; use crate::record::{BoundaryInfo, BoundaryKind, SinkError}; @@ -260,15 +260,38 @@ impl CatalogCapture { } pub async fn capture_boundary( + &self, + info: &BoundaryInfo, + commit_lsn: u64, + parked: &ParkedAt, + ) -> Result<(), SinkError> { + self.capture(info, commit_lsn, parked.lsn(), Some(parked)) + .await + } + + /// Commit boundary capture declined to park for: answered from the log or + /// an empty batch, never a shadow read + pub async fn cover_unparked( + &self, + info: &BoundaryInfo, + commit_lsn: u64, + next_lsn: u64, + ) -> Result<(), SinkError> { + self.capture(info, commit_lsn, next_lsn, None).await + } + + async fn capture( &self, info: &BoundaryInfo, commit_lsn: u64, next_lsn: u64, + parked: Option<&ParkedAt>, ) -> Result<(), SinkError> { use std::sync::atomic::Ordering::Relaxed; let start = std::time::Instant::now(); if let BoundaryKind::Command { writer_xid } = info.kind { - let out = self.capture_command(info, writer_xid, next_lsn).await; + let parked = parked.ok_or_else(|| unparked_read(next_lsn))?; + let out = self.capture_command(info, writer_xid, parked).await; self.stats .capture_nanos .fetch_add(start.elapsed().as_nanos() as u64, Relaxed); @@ -290,7 +313,8 @@ impl CatalogCapture { self.cover_stub(commit_lsn, next_lsn).await?; Vec::new() } else { - self.sql_capture(info, commit_lsn, next_lsn).await? + let parked = parked.ok_or_else(|| unparked_read(next_lsn))?; + self.sql_capture(info, commit_lsn, parked).await? }; if !events.is_empty() { let mut buf = self.buffer.lock().await; @@ -370,15 +394,15 @@ impl CatalogCapture { &self, info: &BoundaryInfo, writer_xid: u32, - next_lsn: u64, + parked: &ParkedAt, ) -> Result<(), SinkError> { use std::sync::atomic::Ordering::Relaxed; + let next_lsn = parked.lsn(); let top_xid = info.drain_xid; let oids: Vec = info.oids.iter().map(|a| a.oid).collect(); let read = { let mut cat = self.catalog.lock().await; - cat.fetch_overlay_descriptors(&oids, top_xid, next_lsn) - .await + cat.fetch_overlay_descriptors(&oids, top_xid, parked).await }; let descs = match read { Ok(descs) => descs, @@ -454,18 +478,19 @@ impl CatalogCapture { &self, info: &BoundaryInfo, commit_lsn: u64, - next_lsn: u64, + parked: &ParkedAt, ) -> Result, SinkError> { use std::sync::atomic::Ordering::Relaxed; + let next_lsn = parked.lsn(); self.stats.sql_captures.fetch_add(1, Relaxed); let (replay_lsn, descs) = { let mut cat = self.catalog.lock().await; if info.capture_all { self.stats.capture_all_runs.fetch_add(1, Relaxed); - cat.fetch_all_descriptors_at(next_lsn).await + cat.fetch_all_descriptors_at(parked).await } else { let oids: Vec = info.oids.iter().map(|a| a.oid).collect(); - cat.fetch_descriptors_batch_at(&oids, next_lsn).await + cat.fetch_descriptors_batch_at(&oids, parked).await } } .map_err(|e| SinkError::Other(format!("descriptor capture at {commit_lsn:#X}: {e}")))?; @@ -719,6 +744,12 @@ impl CatalogCapture { } } +fn unparked_read(next_lsn: u64) -> SinkError { + SinkError::Other(format!( + "catalog boundary at {next_lsn:#X} needs a shadow read but replay was not parked" + )) +} + /// Why an overlay read failed, as the transaction's degrade reason. A /// replay position other than the boundary is its own case: shadow cannot /// rewind, so nothing later re-reads that point @@ -973,13 +1004,25 @@ impl CaptureSet { } pub async fn capture_boundary( + &self, + info: &BoundaryInfo, + commit_lsn: u64, + parked: &ParkedAt, + ) -> Result<(), SinkError> { + for capture in self.scoped(info.db_oid) { + capture.capture_boundary(info, commit_lsn, parked).await?; + } + Ok(()) + } + + pub async fn cover_unparked( &self, info: &BoundaryInfo, commit_lsn: u64, next_lsn: u64, ) -> Result<(), SinkError> { for capture in self.scoped(info.db_oid) { - capture.capture_boundary(info, commit_lsn, next_lsn).await?; + capture.cover_unparked(info, commit_lsn, next_lsn).await?; } Ok(()) } diff --git a/src/source/manifest.rs b/src/source/manifest.rs index b214d768..87a13bfe 100644 --- a/src/source/manifest.rs +++ b/src/source/manifest.rs @@ -60,8 +60,8 @@ use serde::{Deserialize, Serialize}; use thiserror::Error; use crate::pos::{ - Drain, FilterDurable, Floor, LsnKind, Pos, ResumeSafe, ShadowFlush, ShadowReplay, - SourceReceived, Switchpoint, + Drain, Durable, FilterDurable, Floor, LsnKind, Pos, RawStart, ResumeSafe, ShadowFlush, + ShadowReplay, SourceReceived, Switchpoint, }; use crate::record::WAL_SEG_SIZE; use crate::source::wal_stream::WalStream; @@ -186,12 +186,13 @@ pub fn resolved_floor( /// live ack by up to one status interval); slot errors surface at /// START_REPLICATION, same exposure as the boot-scan clamp had. pub fn resolve_start( - raw_start: Pos, + raw_start: Pos, floor: Option>, pinned: bool, archive_end: Option>, shadow: ShadowFloor, ) -> Pos { + // Pinned and greenfield picks floor the same way an acked position does let raw_start: Pos = raw_start.retag(); const UNBOUNDED: u64 = u64::MAX; let inputs = match floor.filter(|f| !f.is_zero()) { @@ -229,11 +230,11 @@ pub fn resolve_start( /// Seed the live pipeline ack from this value, not zero, so the first status /// write cannot discard persisted resume state before WAL re-read catches up pub fn resolve_resume_lsn( - start_lsn: Option>, - bootstrap_end_lsn: Option>, + start_lsn: Option>, + bootstrap_end_lsn: Option>, manifest_ack_lsn: Option>, greenfield_head: Pos, -) -> Pos { +) -> Pos { match (start_lsn, bootstrap_end_lsn, manifest_ack_lsn) { (Some(s), _, _) => s, (None, Some(l), _) => l, @@ -341,10 +342,12 @@ pub async fn load( /// Crash-safe persist; `spill_dir` must already exist /// ([`XactBuffer::new`](crate::xact::xact_buffer::XactBuffer) creates it). -pub async fn write(spill_dir: &Path, m: &Manifest) -> Result<(), ManifestError> { +/// Returned floor is what a crash-now restart resumes from, the only thing +/// pruners may cut against +pub async fn write(spill_dir: &Path, m: &Manifest) -> Result, ManifestError> { let text = toml::to_string(m)?; crate::fs::write_atomic(spill_dir, MANIFEST_FILENAME, text.as_bytes()).await?; - Ok(()) + Ok(Durable::new(m.floor)) } #[cfg(test)] diff --git a/src/source/queueing_record_sink.rs b/src/source/queueing_record_sink.rs index 004d24a3..371a14ec 100644 --- a/src/source/queueing_record_sink.rs +++ b/src/source/queueing_record_sink.rs @@ -355,7 +355,6 @@ impl RecordSink for QueueingRecordSink { next_lsn: record.next_lsn, page_magic: record.page_magic, route: record.route, - catalog_boundary: record.catalog_boundary, boundary_info: record.boundary_info.clone(), aborted_tree: record.aborted_tree.clone(), defer_catalog_decode: record.defer_catalog_decode, diff --git a/src/source/resume_prefix.rs b/src/source/resume_prefix.rs index 7d248b54..30a76afa 100644 --- a/src/source/resume_prefix.rs +++ b/src/source/resume_prefix.rs @@ -274,7 +274,7 @@ mod tests { std::fs::write(segment_path(dir.path(), 1, WAL_SEG_SIZE, start), &retained).unwrap(); for chunk_size in [17, WAL_SEG_SIZE as usize] { - let mut stream = WalStream::new(1, WAL_SEG_SIZE, Pos::new(start)).unwrap(); + let mut stream = WalStream::builder(1, WAL_SEG_SIZE, Pos::new(start)).unwrap(); stream .preserve_resume_prefix(&[dir.path().to_path_buf()]) .await @@ -284,6 +284,7 @@ mod tests { let mut records = CollectingRecordSink::default(); let mut segments = CollectingSegmentSink::default(); let mut lsn = start; + let mut stream = stream.start(); for part in [&raw[..CONT_END], &raw[CONT_END..]] { for chunk in part.chunks(chunk_size) { stream @@ -313,7 +314,7 @@ mod tests { let mut retained = page(start, 100, WAL_SEG_SIZE); retained.resize(WAL_SEG_SIZE as usize, 0); std::fs::write(segment_path(dir.path(), 1, WAL_SEG_SIZE, start), &retained).unwrap(); - let mut stream = WalStream::new(1, WAL_SEG_SIZE, Pos::new(start)).unwrap(); + let mut stream = WalStream::builder(1, WAL_SEG_SIZE, Pos::new(start)).unwrap(); stream .preserve_resume_prefix(&[dir.path().to_path_buf()]) .await @@ -322,6 +323,7 @@ mod tests { retained[XLP_REM_LEN..XLP_REM_LEN + 4].copy_from_slice(&101u32.to_le_bytes()); let mut records = CollectingRecordSink::default(); let mut segments = CollectingSegmentSink::default(); + let mut stream = stream.start(); let error = stream .push(start, &retained, &mut records, &mut segments) .await diff --git a/src/source/segment_sink.rs b/src/source/segment_sink.rs index dc04810a..d5880fbb 100644 --- a/src/source/segment_sink.rs +++ b/src/source/segment_sink.rs @@ -7,11 +7,30 @@ use std::pin::Pin; use tokio::sync::mpsc; use walrus::pg::wal::segment::SegmentName; +use crate::pos::{Durable, FilterDurable, Pos}; use crate::record::{SegmentSink, SinkError}; +/// Renamed segment awaiting filesystem flush pub struct SegFsync { - pub end_lsn: u64, - pub seg_path: PathBuf, + end_lsn: Pos, + seg_path: PathBuf, +} + +impl SegFsync { + pub fn end_lsn(&self) -> Pos { + self.end_lsn + } + + pub fn seg_path(&self) -> &Path { + &self.seg_path + } + + /// `syncfs` the filesystem under `dir`, proving every segment renamed + /// there up to `self` durable + pub fn syncfs(&self, dir: &std::fs::File) -> std::io::Result> { + crate::fs::syncfs(dir)?; + Ok(Durable::new(self.end_lsn)) + } } enum Durability { @@ -59,7 +78,7 @@ impl DirSegmentSink { Durability::Inline => crate::fs::fsync_dir(&self.out_dir).await?, Durability::Background { seg_size, tx } => tx .send(SegFsync { - end_lsn: seg.start_lsn(*seg_size) + bytes.len() as u64, + end_lsn: Pos::new(seg.start_lsn(*seg_size) + bytes.len() as u64), seg_path, }) .await diff --git a/src/source/shadow_stream.rs b/src/source/shadow_stream.rs index 61538fd3..53bfbd04 100644 --- a/src/source/shadow_stream.rs +++ b/src/source/shadow_stream.rs @@ -426,8 +426,9 @@ impl ShadowStreamState { .map(|(id, c)| (*id, c.dispatched_lsn.get(), c.phase.cut())) .collect(); for (id, conn_offset, ends_at) in targets { - let cut = ends_at.map(|s| s.ends_at).filter(|v| *v < end_lsn); - let take = cut.unwrap_or(end_lsn).saturating_sub(start_lsn) as usize; + let cut = ends_at.filter(|s| s.ends_at < end_lsn); + let stop = cut.as_ref().map_or(end_lsn, |s| s.ends_at); + let take = stop.saturating_sub(start_lsn) as usize; let skip = conn_offset.saturating_sub(start_lsn) as usize; let to_send = &bytes[skip.min(take)..take.min(bytes.len())]; let frame_lsn = start_lsn + skip as u64; @@ -436,10 +437,10 @@ impl ShadowStreamState { encode_wal_data_frame_into(out, frame_lsn, server_wal_end, to_send); }) { - self.advance_dispatched(id, Pos::new(cut.unwrap_or(end_lsn))); + self.advance_dispatched(id, Pos::new(stop)); } - if cut.is_some() { - self.end_timeline_for(id, ends_at.expect("cut came from ends_at")); + if let Some(switch) = cut { + self.end_timeline_for(id, switch); } } } diff --git a/src/source/streaming_walker.rs b/src/source/streaming_walker.rs index ac92268c..30037a7a 100644 --- a/src/source/streaming_walker.rs +++ b/src/source/streaming_walker.rs @@ -104,6 +104,11 @@ impl Pending { } } +/// [`StreamingWalker::first_segment_ready`] held, so truncation cannot cut +/// a record whose NOOP rewrite still has to reach seg 0 +#[derive(Debug)] +pub struct FirstSegmentReady(()); + /// Records yield as soon as their last byte arrives pub struct StreamingWalker { seg_size: usize, @@ -141,6 +146,7 @@ impl StreamingWalker { } } + #[cfg(test)] pub fn buffer_len(&self) -> usize { self.buf.len() } @@ -169,23 +175,26 @@ impl StreamingWalker { self.buf.extend_from_slice(bytes); } - /// Gates first-segment flush: pending with `start_offset < seg_size` - /// must complete before the seg ships, so - /// [`rewrite_record`](Self::rewrite_record) applies the NOOP across - /// both segs uniformly. + #[cfg(test)] pub fn pending_start_offset(&self) -> Option { self.pending.as_ref().map(|p| p.start_offset) } + /// Gates first-segment flush: segment whole, and no pending record + /// starts inside it, so [`rewrite_record`](Self::rewrite_record) applies + /// a NOOP across both segs uniformly before seg 0 ships + pub fn first_segment_ready(&self) -> Option { + let spans_out = self + .pending + .as_ref() + .is_some_and(|p| p.start_offset < self.seg_size); + (self.buf.len() >= self.seg_size && !spans_out).then_some(FirstSegmentReady(())) + } + /// Drop the first `seg_size` bytes, rebasing every walker offset by - /// `-seg_size`. Pre: `buf.len() >= seg_size`, any in-flight `pending` - /// lives in `[seg_size, buf.len())`. - pub fn truncate_first_segment(&mut self) { + /// `-seg_size` + pub fn truncate_first_segment(&mut self, _ready: FirstSegmentReady) { let n = self.seg_size; - debug_assert!(self.buf.len() >= n); - if let Some(p) = &self.pending { - debug_assert!(p.start_offset >= n); - } self.buf.drain(0..n); if let Some(p) = self.pending.as_mut() { p.start_offset -= n; @@ -783,8 +792,10 @@ mod tests { ); } - assert!(walker.buffer_len() >= PAGE_SIZE); - walker.truncate_first_segment(); + let ready = walker + .first_segment_ready() + .expect("seg 0 whole, pending past it"); + walker.truncate_first_segment(ready); assert_eq!(walker.buffer_len(), PAGE_SIZE); assert!(walker.pending_start_offset().is_none()); // Sentinel survives at the seg-1 continuation post-truncate. diff --git a/src/source/transition.rs b/src/source/transition.rs index a8a5afed..c48761fa 100644 --- a/src/source/transition.rs +++ b/src/source/transition.rs @@ -292,31 +292,42 @@ impl ForkWait { } } +/// Every consumer has reached the fork at `switch_lsn`, the license +/// [`Switchover::cross`] takes +#[derive(Debug)] +pub struct BarrierOpen { + switch_lsn: Pos, +} + impl ForkBarrier { - /// `None` once every consumer has reached the fork; otherwise what is still + /// Open once every consumer has reached the fork; otherwise what is still /// behind. - pub fn pending(&self, switch_lsn: Pos, seg_size: u64) -> Option { + pub fn open( + &self, + switch_lsn: Pos, + seg_size: u64, + ) -> Result { // One barrier, three roles: retag each term onto barrier's role let fork_segment: Pos = Pos::new(WalStream::align_down(switch_lsn.get(), seg_size)); if self.resume_safe_lsn.retag() < fork_segment { - return Some(ForkWait::EmitterAck { + return Err(ForkWait::EmitterAck { acked: self.resume_safe_lsn, fork_segment, }); } if self.shadow_apply_lsn.is_none_or(|a| a.retag() < switch_lsn) { - return Some(ForkWait::ShadowApply { + return Err(ForkWait::ShadowApply { applied: self.shadow_apply_lsn, fork: switch_lsn, }); } if fork_segment > self.filter_durable.retag().max(self.floor) { - return Some(ForkWait::ArchiveSeal { + return Err(ForkWait::ArchiveSeal { durable: self.filter_durable, fork_segment, }); } - None + Ok(BarrierOpen { switch_lsn }) } } @@ -344,9 +355,23 @@ impl ForkPoint { /// descendant. Nothing published afterwards may leave the floor on the ancestor. #[derive(Debug, Clone, Copy)] pub struct ForkResume { - pub timeline: u32, - pub floor: Pos, - pub switch_lsn: Pos, + timeline: u32, + floor: Pos, + switch_lsn: Pos, +} + +impl ForkResume { + pub fn timeline(&self) -> u32 { + self.timeline + } + + pub fn floor(&self) -> Pos { + self.floor + } + + pub fn switch_lsn(&self) -> Pos { + self.switch_lsn + } } /// Where a crossing read descendant's copy of fork prefix @@ -495,9 +520,8 @@ impl Switchover<'_> { /// live timeline; each call crosses exactly one fork. /// /// `commit` persists [`ForkResume`] and must not return until it is durable: - /// it is the point the stream stops being the ancestor's. The caller runs - /// the barrier before getting here, so by now nothing below the fork is in - /// flight anywhere. + /// it is the point the stream stops being the ancestor's. `open` proves + /// nothing below the fork is in flight anywhere. /// /// `slot` comes per call, not off the struct: it reloads with `[source]`, /// so a crossing after a repoint asks the descendant for the slot the @@ -513,12 +537,18 @@ impl Switchover<'_> { status: StandbyStatus, guards: ForkGuards, fork: &ForkPoint, + open: BarrierOpen, commit: C, stats: &mut TimelineStats, ) -> Result where C: AsyncFnOnce(ForkResume) -> anyhow::Result<()>, { + debug_assert_eq!( + open.switch_lsn.get(), + fork.switch_lsn, + "barrier for another fork" + ); let started = std::time::Instant::now(); let out = self .cross_inner( @@ -1202,7 +1232,7 @@ mod tests { #[test] fn barrier_opens_once_every_frontier_reaches_the_fork() { - assert_eq!(caught_up().pending(Pos::new(SEG + 0x1000), SEG), None); + assert_eq!(caught_up().open(Pos::new(SEG + 0x1000), SEG).err(), None); } /// An unacked commit *inside* the fork segment does not hold the barrier: a @@ -1215,7 +1245,8 @@ mod tests { resume_safe_lsn: SEG.into(), ..caught_up() } - .pending(Pos::new(SEG + 0x1000), SEG), + .open(Pos::new(SEG + 0x1000), SEG) + .err(), None, "acked exactly at the fork segment's start is enough", ); @@ -1224,7 +1255,8 @@ mod tests { resume_safe_lsn: (SEG - 1).into(), ..caught_up() } - .pending(Pos::new(SEG + 0x1000), SEG), + .open(Pos::new(SEG + 0x1000), SEG) + .err(), Some(ForkWait::EmitterAck { acked: (SEG - 1).into(), fork_segment: SEG.into(), @@ -1241,7 +1273,7 @@ mod tests { ..caught_up() }; assert_eq!( - b.pending(fork, SEG), + b.open(fork, SEG).err(), Some(ForkWait::ShadowApply { applied, fork }), "applied {applied:?} must not open the barrier", ); @@ -1258,7 +1290,7 @@ mod tests { ..caught_up() }; assert_eq!( - b.pending(Pos::new(SEG + 0x1000), SEG), + b.open(Pos::new(SEG + 0x1000), SEG).err(), Some(ForkWait::ArchiveSeal { durable: Pos::ZERO, fork_segment: SEG.into(), @@ -1269,7 +1301,8 @@ mod tests { floor: SEG.into(), ..b } - .pending(Pos::new(SEG + 0x1000), SEG), + .open(Pos::new(SEG + 0x1000), SEG) + .err(), None, "a floor already at the fork segment needs no seal", ); diff --git a/src/source/wal_stream.rs b/src/source/wal_stream.rs index 2175820f..1b78f5cd 100644 --- a/src/source/wal_stream.rs +++ b/src/source/wal_stream.rs @@ -132,6 +132,46 @@ pub struct WalStream { resume_prefix: Option, } +/// [`WalStream`] before its first push. Swapping the bytes sink mid-stream +/// would leave bytes already dispatched unreceived by the new sink, and filter +/// scope set late would misroute records already decided +pub struct WalStreamBuilder(WalStream); + +impl WalStreamBuilder { + /// Preserve continuation of a record already rewritten before resume. + /// `dirs` rank by trust: the first holding the resume segment wins + pub async fn preserve_resume_prefix(&mut self, dirs: &[PathBuf]) -> Result<(), WalStreamError> { + let s = &mut self.0; + s.resume_prefix = ResumePrefix::load(dirs, s.timeline, s.seg_size, s.next_lsn) + .await + .map_err(WalStreamError::ResumePrefix)?; + if let Some(prefix) = &s.resume_prefix { + tracing::info!( + start_lsn = format_args!("{:#X}", s.next_lsn), + end_lsn = format_args!("{:#X}", prefix.end_lsn()), + "preserving filtered resume continuation" + ); + } + Ok(()) + } + + pub fn filter(&self) -> &Filter { + &self.0.filter + } + + pub fn filter_mut(&mut self) -> &mut Filter { + &mut self.0.filter + } + + pub fn set_bytes_sink(&mut self, sink: Box) { + self.0.bytes_sink = sink; + } + + pub fn start(self) -> WalStream { + self.0 + } +} + /// Digest of the ancestor bytes a descendant repeats, from /// [`WalStream::fork_prefix`]. #[derive(Debug, Clone, Copy, PartialEq, Eq)] @@ -169,20 +209,14 @@ impl WalStream { }) } - /// Preserve continuation of a record already rewritten before resume. - /// `dirs` rank by trust: the first holding the resume segment wins - pub async fn preserve_resume_prefix(&mut self, dirs: &[PathBuf]) -> Result<(), WalStreamError> { - self.resume_prefix = ResumePrefix::load(dirs, self.timeline, self.seg_size, self.next_lsn) - .await - .map_err(WalStreamError::ResumePrefix)?; - if let Some(prefix) = &self.resume_prefix { - tracing::info!( - start_lsn = format_args!("{:#X}", self.next_lsn), - end_lsn = format_args!("{:#X}", prefix.end_lsn()), - "preserving filtered resume continuation" - ); - } - Ok(()) + /// Stream that needs a bytes sink, resume prefix, or filter scope, all of + /// which are fixed before the first [`push`](Self::push) + pub fn builder( + timeline: u32, + seg_size: u64, + start_lsn: Pos, + ) -> Result { + Self::new(timeline, seg_size, start_lsn).map(WalStreamBuilder) } /// Stats here are cumulative across every segment this stream processed. @@ -190,17 +224,6 @@ impl WalStream { &self.filter } - pub fn filter_mut(&mut self) -> &mut Filter { - &mut self.filter - } - - /// Must be called before the first [`push`](Self::push); swapping - /// mid-stream leaves bytes already dispatched to the prior sink - /// unreceived by the new one. - pub fn set_bytes_sink(&mut self, sink: Box) { - self.bytes_sink = sink; - } - pub fn align_down(lsn: u64, seg_size: u64) -> u64 { lsn - (lsn % seg_size) } @@ -381,7 +404,6 @@ impl WalStream { next_lsn, page_magic, route, - catalog_boundary: verdict.catalog_boundary, boundary_info: verdict.boundary, aborted_tree: verdict.aborted_tree, defer_catalog_decode: verdict.defer_catalog_decode, @@ -403,14 +425,9 @@ impl WalStream { segment_sink: &mut (dyn SegmentSink + Send), ) -> Result { let seg_size = self.seg_size as usize; - if self.walker.buffer_len() < seg_size { - return Ok(false); - } - if let Some(pend_off) = self.walker.pending_start_offset() - && pend_off < seg_size - { + let Some(ready) = self.walker.first_segment_ready() else { return Ok(false); - } + }; let seg = self.segment_for_lsn(self.current_lsn); // Wire tail: if wire_offset < seg_size, the residual span // (alignment pad + page header + in-place-rewritten spanning bytes) @@ -425,7 +442,7 @@ impl WalStream { segment_sink .on_segment(seg, &self.walker.buffer()[..seg_size]) .await?; - self.walker.truncate_first_segment(); + self.walker.truncate_first_segment(ready); self.current_lsn += self.seg_size; self.bytes_sink.on_segment_retired(self.current_lsn).await?; self.wire_offset = self.wire_offset.saturating_sub(seg_size); @@ -838,10 +855,10 @@ mod tests { assert_eq!(std::fs::read(&seg_path).unwrap(), bytes); let msg = rx.try_recv().expect("segment enqueued for fsync"); assert_eq!( - msg.end_lsn, + msg.end_lsn(), seg.start_lsn(WAL_SEG_SIZE) + bytes.len() as u64 ); - assert_eq!(msg.seg_path, seg_path); + assert_eq!(msg.seg_path(), seg_path); } /// Contract: a `RecordBytesSink` sees the full wire stream (record @@ -870,8 +887,9 @@ mod tests { } let page = synth_two_record_page(); - let mut ws = WalStream::new(1, SEG, Pos::ZERO).unwrap(); + let mut ws = WalStream::builder(1, SEG, Pos::ZERO).unwrap(); ws.set_bytes_sink(Box::new(SharedCollector(collector_chunks.clone()))); + let mut ws = ws.start(); ws.push(0, &page, &mut rec, &mut seg).await.unwrap(); assert!(!rec.records.is_empty(), "record_sink fired"); @@ -1082,7 +1100,7 @@ mod tests { let page = page_of(&[raw_rec(RmId::Smgr as u8, 0x10, 0, None, Some(&md))], 8192); for fail_persist in [false, true] { let tmp = tempfile::tempdir().unwrap(); - let mut stream = WalStream::new(1, 8192, Pos::new(0u64)).unwrap(); + let mut stream = WalStream::builder(1, 8192, Pos::new(0u64)).unwrap(); stream.filter_mut().keep_user_rels(Default::default(), 0); stream .filter_mut() @@ -1098,6 +1116,7 @@ mod tests { stream.set_bytes_sink(Box::new(RejectWire(calls.clone()))); let mut records = CollectingRecordSink::default(); let mut segments = CollectingSegmentSink::default(); + let mut stream = stream.start(); let err = stream .push(0, &page, &mut records, &mut segments) .await @@ -1134,10 +1153,11 @@ mod tests { // User heap insert in the followed db: dropped, so NOOP-rewritten let user = raw_rec(RmId::Heap as u8, 0x00, 8, Some((1663, 5, 50000)), None); let page = page_of(&[user], 8192); - let mut ws = WalStream::new(1, SEG, Pos::ZERO).unwrap(); + let mut ws = WalStream::builder(1, SEG, Pos::ZERO).unwrap(); ws.filter_mut().set_target_db(5); let mut rec = CollectingRecordSink::default(); let mut seg = CollectingSegmentSink::default(); + let mut ws = ws.start(); ws.push(0, &page, &mut rec, &mut seg).await.unwrap(); let prefix = ws.fork_prefix(); @@ -1449,7 +1469,7 @@ mod tests { } } - let mut ws = WalStream::new(1, SEG, Pos::ZERO).unwrap(); + let mut ws = WalStream::builder(1, SEG, Pos::ZERO).unwrap(); // Live wiring: catalog dirt is admitted only for the followed db ws.filter_mut().set_target_db(5); ws.set_bytes_sink(Box::new(SpanLog(chunks.clone()))); @@ -1463,6 +1483,7 @@ mod tests { ); let stats = gate.stats.clone(); let mut sink = BoundaryHoldSink::new(q, gate); + let mut ws = ws.start(); let pump = tokio::spawn(async move { let mut seg = CollectingSegmentSink::default(); ws.push(0, &page, &mut sink, &mut seg).await.unwrap(); @@ -1578,12 +1599,13 @@ mod tests { 0, 64 * 1024 * 1024, ))); - let mut ws = WalStream::new(1, SEG, Pos::ZERO).unwrap(); + let mut ws = WalStream::builder(1, SEG, Pos::ZERO).unwrap(); ws.set_bytes_sink(Box::new(ShadowStreamSink::new(state.clone()))); let mut rec = CollectingRecordSink::default(); let mut seg = CollectingSegmentSink::default(); let (page0, page1) = synth_two_page_spanning_record(); + let mut ws = ws.start(); ws.push(0, &page0, &mut rec, &mut seg).await.unwrap(); ws.push(SEG, &page1, &mut rec, &mut seg).await.unwrap(); diff --git a/src/source_db.rs b/src/source_db.rs index 806d29f9..7b7e212e 100644 --- a/src/source_db.rs +++ b/src/source_db.rs @@ -73,8 +73,8 @@ impl DbLink { socket = %bridge.path().display(), dbname = %cfg.name, workers = bridge.pool_size(), - pg_version = info.map(|i| i.pg_version_num).unwrap_or(0), - in_recovery = info.map(|i| i.in_recovery).unwrap_or(false), + pg_version = info.pg_version_num, + in_recovery = info.in_recovery, "bridge connected", ); let cat_cfg = ShadowCatalogConfig::default(); @@ -92,16 +92,14 @@ impl DbLink { .current_database_oid() .await .with_context(|| format!("shadow database oid for {}", cfg.name))?; - if let Some(hello) = info { - anyhow::ensure!( - hello.datid == oid, - "bridge socket {} serves database oid {}, but {} is oid {oid}; \ - shadow's walshadow.databases disagrees with this config", - bridge.path().display(), - hello.datid, - cfg.name, - ); - } + anyhow::ensure!( + info.datid == oid, + "bridge socket {} serves database oid {}, but {} is oid {oid}; \ + shadow's walshadow.databases disagrees with this config", + bridge.path().display(), + info.datid, + cfg.name, + ); tracing::info!( target: "walshadow", conninfo = %cfg.shadow_conninfo, diff --git a/src/toast/resolver.rs b/src/toast/resolver.rs index cbe9b775..182ab8d6 100644 --- a/src/toast/resolver.rs +++ b/src/toast/resolver.rs @@ -721,7 +721,8 @@ impl ClickHouseChunkStore { client.send_data(Some(bb)).await?; client.send_data_end().await?; } - drain_to_end_of_stream(&mut client).await + drain_to_end_of_stream(&mut client).await?; + Ok(()) }) .await; (client, result) diff --git a/src/toast/shadow_store.rs b/src/toast/shadow_store.rs index b9ec3b9c..04d4a8a6 100644 --- a/src/toast/shadow_store.rs +++ b/src/toast/shadow_store.rs @@ -122,7 +122,7 @@ impl ShadowToastStore { /// stalled replay and restart timeout whenever replay position changes async fn await_replay(&self, bridge: &Bridge, through: u64) -> Result { // Primary files are complete before service and have no replay position - if through == 0 || !bridge.info().is_some_and(|i| i.in_recovery) { + if through == 0 || !bridge.info().in_recovery { return Ok(0); } let mut since = Instant::now(); diff --git a/src/toast/toast_retire.rs b/src/toast/toast_retire.rs index a3997821..a23a962a 100644 --- a/src/toast/toast_retire.rs +++ b/src/toast/toast_retire.rs @@ -130,6 +130,7 @@ impl RetireLedger { self.entries .iter() .copied() + // Restart resumes at the floor, so a commit below it never replays .filter(|&(_, commit_lsn)| commit_lsn.retag() < cut) .collect() } diff --git a/src/xact/xact_buffer.rs b/src/xact/xact_buffer.rs index f69d7f78..a3c3f8cb 100644 --- a/src/xact/xact_buffer.rs +++ b/src/xact/xact_buffer.rs @@ -259,9 +259,9 @@ pub enum XactBufferError { from_lsn: u64, through_lsn: u64, }, - /// Raw entries drained with no commit-time resolution installed: the + /// Raw entries drained under [`StashResolved::nothing_stashed`]: the /// discard arm would swallow every stashed row, fence included. Fail - /// closed — resolve_stash must run for any xact that stashed + /// closed, resolve_stash must run for any xact that stashed #[error("drain for xact {top_xid} has stashed records but no resolution")] MissingStashResolution { top_xid: u32 }, /// Merged entry carries a writer xid outside the owning xact + @@ -582,9 +582,24 @@ pub struct StashResolution { stats: Option>, } +/// Commit-time verdict [`XactBuffer::drain_committed`] consumes, so a drain +/// cannot run against an unresolved tree or another xid's resolution +pub struct StashResolved { + top_xid: u32, + /// `None` asserts the tree stashed nothing, checked at drain + res: Option, +} + +impl StashResolved { + /// Drain without resolving; fails closed if the tree stashed after all + pub const fn nothing_stashed(top_xid: u32) -> Self { + Self { top_xid, res: None } + } +} + /// Resolve the finishing tree's stashed filenodes against the descriptor /// log at the commit's `next_lsn` (capture ran inside the boundary hold, so -/// same-xact CREATE/rewrite descriptors are already covered), install +/// same-xact CREATE/rewrite descriptors are already covered), return /// outcomes for the imminent drain, and queue `O - B` barriers for /// marker-proven toast generations. A toast heap without its marker fails /// closed ([`XactBufferError::IncompleteToastGeneration`]). @@ -605,7 +620,7 @@ pub async fn resolve_stash( subxids: &[u32], next_lsn: u64, stats: Arc, -) -> std::result::Result<(), XactBufferError> { +) -> std::result::Result { let rfns = { let buf = buffer.lock().await; let mut xids: Vec = Vec::with_capacity(1 + subxids.len()); @@ -614,7 +629,10 @@ pub async fn resolve_stash( buf.stash_candidates(&xids) }; if rfns.is_empty() { - return Ok(()); + return Ok(StashResolved { + top_xid, + res: Some(StashResolution::default()), + }); } let mut outcomes: HashMap = HashMap::with_capacity(rfns.len()); let mut barriers: Vec<(u32, u64)> = Vec::with_capacity(rfns.len()); @@ -705,14 +723,13 @@ pub async fn resolve_stash( } let resolved: Vec = rfns.iter().map(|(rfn, _)| *rfn).collect(); buf.forget_markers(&resolved); - buf.install_stash_resolution( + Ok(StashResolved { top_xid, - StashResolution { + res: Some(StashResolution { outcomes, stats: Some(stats), - }, - ); - Ok(()) + }), + }) } /// Whether `rfn` replaced a different live filenode for `oid`, read before @@ -741,9 +758,6 @@ pub struct XactBuffer { /// `markers` (consumed out-of-band, or filenode reused by a later /// generation holding its own queue entry) marker_order: VecDeque<(RelFileNode, u64)>, - /// Commit-time resolution installed by [`resolve_stash`] just before - /// the drain pops it, keyed by top xid - pending_stash: HashMap, /// Committed transactions waiting for durable acknowledgment pending_durable: PendingDurable, bytes_in_memory: usize, @@ -782,7 +796,6 @@ impl XactBuffer { inflight: HashMap::new(), markers: HashMap::new(), marker_order: VecDeque::new(), - pending_stash: HashMap::new(), pending_durable: PendingDurable::default(), bytes_in_memory: 0, stats: XactBufferStats::default(), @@ -1085,11 +1098,6 @@ impl XactBuffer { } } - /// Install commit-time resolution for `top_xid`'s imminent drain - pub fn install_stash_resolution(&mut self, top_xid: u32, res: StashResolution) { - self.pending_stash.insert(top_xid, res); - } - /// Drains in `source_lsn` order at commit, so a DDL's `Added`/`Changed` /// event lands BEFORE the heap writes that follow it pub fn on_schema_event(&mut self, xid: u32, source_lsn: u64, event: SchemaEvent) { @@ -1158,10 +1166,12 @@ impl XactBuffer { async fn evict_xact(&mut self, xid: u32) -> std::result::Result<(), XactBufferError> { let st = self.inflight.get_mut(&xid).expect("xid present"); let first_spill = st.spill.is_none(); - if first_spill { - st.spill = Some(self.store.writer(xid, st.first_lsn.get()).await?); - } - let writer = st.spill.as_mut().unwrap(); + let writer = match &mut st.spill { + Some(writer) => writer, + None => st + .spill + .insert(self.store.writer(xid, st.first_lsn.get()).await?), + }; let drained: Vec = std::mem::take(&mut st.in_mem); let freed = std::mem::take(&mut st.in_mem_bytes); writer.write_batch(drained).await?; @@ -1264,7 +1274,7 @@ impl XactBuffer { /// owns `emitter_ack_lsn`. pub async fn drain_committed( &mut self, - top_xid: u32, + stash: StashResolved, commit_ts: i64, commit_lsn: u64, subxids: &[u32], @@ -1272,6 +1282,7 @@ impl XactBuffer { // when a put consumer exists collect_rows: bool, ) -> std::result::Result { + let top_xid = stash.top_xid; let mut xids: Vec = Vec::with_capacity(1 + subxids.len()); xids.push(top_xid); xids.extend_from_slice(subxids); @@ -1289,7 +1300,6 @@ impl XactBuffer { return Ok(CommittedDrain { commit_ts, commit_lsn, - had_states: false, merged: None, generations: Vec::new(), }); @@ -1304,10 +1314,8 @@ impl XactBuffer { } self.stats.bytes_in_memory = self.bytes_in_memory as u64; // Absent resolution would send every Raw entry through fold_raw's - // discard arm — fence included. Resolution is installed by - // resolve_stash for any tree that stashed, so absence is a wiring - // bug, not a verdict - let stash = match self.pending_stash.remove(&top_xid) { + // discard arm, fence included + let stash = match stash.res { Some(res) => res, None if states.iter().all(|st| st.stash_rfns.is_empty()) => StashResolution::default(), None => return Err(XactBufferError::MissingStashResolution { top_xid }), @@ -1327,7 +1335,6 @@ impl XactBuffer { Ok(CommittedDrain { commit_ts, commit_lsn, - had_states: true, merged: Some(merged), generations: Vec::new(), }) @@ -1393,10 +1400,6 @@ impl XactBuffer { for rfn in st.stash_rfns.keys() { self.markers.remove(rfn); } - // A resolution installed for a xid that then aborted must not - // outlive it: the next xact reusing the xid would fold its raws - // under a foreign descriptor and a foreign fence - self.pending_stash.remove(&x); any = true; self.stats.xacts_active = self.stats.xacts_active.saturating_sub(1); self.bytes_in_memory = self.bytes_in_memory.saturating_sub(st.in_mem_bytes); @@ -1508,16 +1511,21 @@ impl Drop for PendingGauge { /// Lazy merge source for one xid: spill-reader head (older in WAL order) /// chained with the in-mem tail, one decoded entry resident. `pop` refills -/// from the reader until EOF, then drains `in_mem`. The EOF reader parks in -/// `spent` so the file unlinks only at [`MergedDrain::finish`], +/// from the reader until EOF, then drains `in_mem`. The EOF reader parks as +/// `Spent` so the file unlinks only at [`MergedDrain::finish`], /// post-dispatch. struct MergeSource { head: Option, - reader: Option, - spent: Option, + spill: SpillSide, in_mem: VecDeque, } +enum SpillSide { + Unspilled, + Reading(SpillReader), + Spent(SpillReader), +} + impl MergeSource { async fn open( reader: Option, @@ -1534,27 +1542,29 @@ impl MergeSource { } let mut src = Self { head: None, - reader, - spent: None, + spill: reader.map_or(SpillSide::Unspilled, SpillSide::Reading), in_mem: in_mem.into(), }; - src.refill(gauge).await?; + src.head = src.next_entry(gauge).await?; Ok(src) } - async fn refill(&mut self, gauge: &mut ResidentGauge) -> std::result::Result<(), SpillError> { - debug_assert!(self.head.is_none()); - if let Some(r) = self.reader.as_mut() { + async fn next_entry( + &mut self, + gauge: &mut ResidentGauge, + ) -> std::result::Result, SpillError> { + if let SpillSide::Reading(r) = &mut self.spill { if let Some(entry) = r.next().await? { gauge.add(approximate_size(&entry)); - self.head = Some(entry); - return Ok(()); + return Ok(Some(entry)); + } + if let SpillSide::Reading(r) = std::mem::replace(&mut self.spill, SpillSide::Unspilled) + { + self.spill = SpillSide::Spent(r); } - self.spent = self.reader.take(); } // Already counted at `open`; moving to head keeps it resident - self.head = self.in_mem.pop_front(); - Ok(()) + Ok(self.in_mem.pop_front()) } fn head_lsn(&self) -> Option { @@ -1569,7 +1579,7 @@ impl MergeSource { return Ok(None); }; gauge.sub(approximate_size(&entry)); - self.refill(gauge).await?; + self.head = self.next_entry(gauge).await?; Ok(Some(entry)) } } @@ -1736,19 +1746,19 @@ impl MergedDrain { /// appended once to the body spool past it (resolution map and mirror /// rows share either form) fn fold_body(&mut self, data: &bytes::Bytes) -> std::result::Result { - if self.spool.is_none() { - if self.mem_body_bytes + data.len() <= self.body_mem_max { + let spool = match &mut self.spool { + Some(spool) => spool, + None if self.mem_body_bytes + data.len() <= self.body_mem_max => { self.mem_body_bytes += data.len(); return Ok(Body::Mem(data.clone())); } - self.spool = Some(BodySpoolWriter::create( + None => self.spool.insert(BodySpoolWriter::create( &self.spool_dir, self.spool_xid, self.spool_lsn, Some(self.spool_gauge.clone()), - )?); - } - let spool = self.spool.as_mut().expect("just created"); + )?), + }; let r = spool.append(data)?; Ok(Body::File(r)) } @@ -2023,10 +2033,7 @@ impl MergedDrain { s.unlink()?; } for s in self.sources { - if let Some(r) = s.spent { - r.unlink().await?; - } - if let Some(r) = s.reader { + if let SpillSide::Reading(r) | SpillSide::Spent(r) = s.spill { r.unlink().await?; } } @@ -2125,7 +2132,7 @@ pub struct DrainedBatch { /// `new_rows` cursor for each TRUNCATE heap pub truncate_rows: Vec, /// Last slice of the commit. Only its seq may publish `commit_lsn` in - /// the ack (`register` vs `register_partial`): an earlier slice + /// the ack (`Publish::Commit` vs `Partial`): an earlier slice /// publishing would claim durability for rows still in flight. pub is_final: bool, } @@ -2138,7 +2145,11 @@ pub enum WalkStep { upto: usize, }, Event(DrainEntry), - Truncate(DescribedHeap), + /// Follows the `Rows` step sealing `new_rows[..upto]` + Truncate { + heap: DescribedHeap, + upto: usize, + }, Heap(DescribedHeap), } @@ -2180,7 +2191,7 @@ impl DrainedBatch { .next() .expect("truncate_rows cursor per Truncate heap"); steps.push(WalkStep::Rows { upto }); - steps.push(WalkStep::Truncate(heap)); + steps.push(WalkStep::Truncate { heap, upto }); } else { steps.push(WalkStep::Heap(heap)); } @@ -2206,13 +2217,16 @@ impl DrainedBatch { pub struct CommittedDrain { pub commit_ts: i64, pub commit_lsn: u64, - /// False for read-only / filter-dropped / unknown xid. - pub had_states: bool, + /// `None` for read-only / filter-dropped / unknown xid merged: Option, generations: Vec>, } impl CommittedDrain { + pub fn had_states(&self) -> bool { + self.merged.is_some() + } + /// Next slice, `None` once exhausted. The slice closes at the first /// heap reaching `max_rows` / `max_bytes` (budget is a trigger, not a /// hard cap: one oversized row still ships alone). Slices only cut at @@ -3356,31 +3370,28 @@ pub(crate) mod raw_fixtures { /// `Ordinary` verdict for top xid 1, injected directly (no descriptor /// log in these fixtures) pub(crate) fn inject_ordinary( - b: &mut XactBuffer, rfn: RelFileNode, rel: Arc, - ) { - inject_ordinary_with_stats(b, rfn, rel, None); + ) -> StashResolved { + inject_ordinary_with_stats(rfn, rel, None) } pub(crate) fn inject_ordinary_with_stats( - b: &mut XactBuffer, rfn: RelFileNode, rel: Arc, stats: Option>, - ) { - inject_ordinary_fenced(b, rfn, rel, stats, Vec::new()); + ) -> StashResolved { + inject_ordinary_fenced(rfn, rel, stats, Vec::new()) } /// `Ordinary` verdict carrying a fence, as `resolve_stash` would attach /// it for a filenode with overlapping ambiguity intervals pub(crate) fn inject_ordinary_fenced( - b: &mut XactBuffer, rfn: RelFileNode, rel: Arc, stats: Option>, fence: Vec>, - ) { + ) -> StashResolved { let mut outcomes = HashMap::new(); outcomes.insert( rfn, @@ -3391,18 +3402,19 @@ pub(crate) mod raw_fixtures { pending: Vec::new(), }, ); - b.pending_stash - .insert(1, StashResolution { outcomes, stats }); + StashResolved { + top_xid: 1, + res: Some(StashResolution { outcomes, stats }), + } } /// `Ordinary` verdict carrying this xact's own command-boundary shapes, /// as `resolve_stash` reads them off the pending catalog pub(crate) fn inject_ordinary_pending( - b: &mut XactBuffer, rfn: RelFileNode, rel: Arc, pending: Vec, - ) { + ) -> StashResolved { let mut outcomes = HashMap::new(); outcomes.insert( rfn, @@ -3413,13 +3425,13 @@ pub(crate) mod raw_fixtures { pending, }, ); - b.pending_stash.insert( - 1, - StashResolution { + StashResolved { + top_xid: 1, + res: Some(StashResolution { outcomes, stats: None, - }, - ); + }), + } } /// `[from, through)` interval over one filenode, the shape a physical @@ -3665,7 +3677,7 @@ mod tests { .any(|s| s.spill.is_some() && !s.in_mem.is_empty()) ); let mut drain = buffer - .drain_committed(7, 0, 100, &[8], false) + .drain_committed(StashResolved::nothing_stashed(7), 0, 100, &[8], false) .await .unwrap(); assert_eq!(buffer.stats().bytes_in_memory, 0); @@ -3727,8 +3739,11 @@ mod tests { b.on_heap(heap_with_value(8, 150, 16)).await.unwrap(); assert_eq!(b.resume_safe_lsn(Pos::new(500)), 100); // Keep floor while committed slices remain undurable - let drain = b.drain_committed(8, 0, 200, &[], false).await.unwrap(); - assert!(drain.had_states); + let drain = b + .drain_committed(StashResolved::nothing_stashed(8), 0, 200, &[], false) + .await + .unwrap(); + assert!(drain.had_states()); drop(drain); assert_eq!( b.resume_safe_lsn(Pos::new(180)), @@ -3744,8 +3759,11 @@ mod tests { // Drop floor once acknowledgment reaches commit assert_eq!(b.resume_safe_lsn(Pos::new(200)), 200); // Ignore commits without buffered rows - let empty = b.drain_committed(9, 0, 300, &[], false).await.unwrap(); - assert!(!empty.had_states); + let empty = b + .drain_committed(StashResolved::nothing_stashed(9), 0, 300, &[], false) + .await + .unwrap(); + assert!(!empty.had_states()); assert_eq!(b.resume_safe_lsn(Pos::new(300)), 300); } @@ -3756,7 +3774,10 @@ mod tests { // Use earliest record across transaction tree b.on_heap(heap_with_value(20, 400, 16)).await.unwrap(); b.on_heap(heap_with_value(21, 420, 16)).await.unwrap(); - let drain = b.drain_committed(21, 0, 450, &[20], false).await.unwrap(); + let drain = b + .drain_committed(StashResolved::nothing_stashed(21), 0, 450, &[20], false) + .await + .unwrap(); drop(drain); assert_eq!(b.resume_safe_lsn(Pos::new(440)), 400); assert_eq!(b.resume_safe_lsn(Pos::new(450)), 450); @@ -4624,7 +4645,10 @@ mod tests { // Arrival order 150, 100: 100 must still precede the heap@120 b.on_schema_event(1, 150, dropped_event(9)); b.on_schema_event(1, 100, dropped_event(7)); - let mut drain = b.drain_committed(1, 42, 0x2000, &[], false).await.unwrap(); + let mut drain = b + .drain_committed(StashResolved::nothing_stashed(1), 42, 0x2000, &[], false) + .await + .unwrap(); let mut order: Vec = Vec::new(); while let Some(batch) = drain.next_batch(8, usize::MAX, None).await.unwrap() { order.extend(flatten_batch(&batch)); @@ -4649,9 +4673,12 @@ mod tests { .unwrap(); b.on_heap(heap_with_value(1, 200, 16)).await.unwrap(); b.on_schema_event(1, 100, dropped_event(7)); - inject_ordinary(&mut b, rfn, rel); + let stash = inject_ordinary(rfn, rel); - let mut drain = b.drain_committed(1, 42, 0x2000, &[8], false).await.unwrap(); + let mut drain = b + .drain_committed(stash, 42, 0x2000, &[8], false) + .await + .unwrap(); let batch = drain .next_batch(8, usize::MAX, None) .await @@ -4688,8 +4715,11 @@ mod tests { .await .unwrap(); let stats = Arc::new(EmitterStats::default()); - inject_ordinary_with_stats(&mut b, rfn, rel, Some(stats.clone())); - let mut drain = b.drain_committed(1, 42, 0x2000, &[], false).await.unwrap(); + let stash = inject_ordinary_with_stats(rfn, rel, Some(stats.clone())); + let mut drain = b + .drain_committed(stash, 42, 0x2000, &[], false) + .await + .unwrap(); let batch = drain .next_batch(2, usize::MAX, None) .await @@ -4724,8 +4754,11 @@ mod tests { let mut raw = multi_insert_raw(1, 100, 16411, &[1]); raw.main_data[0] = 0; b.stash_raw(1, raw).await.unwrap(); - inject_ordinary(&mut b, rfn, rel); - let mut drain = b.drain_committed(1, 42, 0x2000, &[], false).await.unwrap(); + let stash = inject_ordinary(rfn, rel); + let mut drain = b + .drain_committed(stash, 42, 0x2000, &[], false) + .await + .unwrap(); let err = drain .next_batch(8, usize::MAX, None) .await @@ -4757,8 +4790,11 @@ mod tests { b.stash_raw(1, multi_insert_raw(1, 150, 16420, &[1])) .await .unwrap(); - inject_ordinary_fenced(&mut b, rfn, rel, None, vec![rfn_ambiguity(rfn, 100, 200)]); - let mut drain = b.drain_committed(1, 42, 0x2000, &[], false).await.unwrap(); + let stash = inject_ordinary_fenced(rfn, rel, None, vec![rfn_ambiguity(rfn, 100, 200)]); + let mut drain = b + .drain_committed(stash, 42, 0x2000, &[], false) + .await + .unwrap(); let err = drain .next_batch(8, usize::MAX, None) .await @@ -4834,7 +4870,7 @@ mod tests { .stash_raw(1, multi_insert_raw(1, 0x200, 16430, &[1])) .await .unwrap(); - resolve_stash( + let stash = resolve_stash( &buffer, &log, &PendingCatalog::default(), @@ -4846,7 +4882,10 @@ mod tests { .await .unwrap(); let mut b = buffer.lock().await; - let mut drain = b.drain_committed(1, 42, 0x2F0, &[], false).await.unwrap(); + let mut drain = b + .drain_committed(stash, 42, 0x2F0, &[], false) + .await + .unwrap(); let err = drain .next_batch(8, usize::MAX, None) .await @@ -4961,14 +5000,16 @@ mod tests { b.stash_raw(1, multi_insert_raw(1, 200, 16421, &[7, 8])) .await .unwrap(); - inject_ordinary_fenced( - &mut b, + let stash = inject_ordinary_fenced( rfn, rel, None, vec![rfn_ambiguity(rfn, 100, 200), rfn_ambiguity(rfn, 20, 40)], ); - let mut drain = b.drain_committed(1, 42, 0x2000, &[], false).await.unwrap(); + let mut drain = b + .drain_committed(stash, 42, 0x2000, &[], false) + .await + .unwrap(); let batch = drain .next_batch(8, usize::MAX, None) .await @@ -4988,7 +5029,7 @@ mod tests { .await .unwrap(); let err = b - .drain_committed(1, 42, 0x2000, &[], false) + .drain_committed(StashResolved::nothing_stashed(1), 42, 0x2000, &[], false) .await .err() .expect("fail-closed with no resolution installed"); @@ -4998,22 +5039,6 @@ mod tests { ); } - /// An aborted tree's resolution must not linger for the next xact that - /// reuses the xid: it would fold raws under a foreign descriptor + fence - #[tokio::test(flavor = "current_thread")] - async fn abort_drops_installed_resolution() { - let tmp = tempdir().unwrap(); - let mut b = XactBuffer::new(cfg(tmp.path().to_path_buf())).unwrap(); - let rel = int4_descriptor(16423); - let rfn = rel.rfn; - b.stash_raw(1, multi_insert_raw(1, 100, 16423, &[1])) - .await - .unwrap(); - inject_ordinary(&mut b, rfn, rel); - b.abort(1, Pos::new(0x1000), &[]).await.unwrap(); - assert!(b.pending_stash.is_empty()); - } - /// INPLACE mutates tuple bytes with no decode shape: typed reject, not /// a silent skip #[tokio::test(flavor = "current_thread")] @@ -5026,8 +5051,11 @@ mod tests { raw.rm = RmId::Heap as u8; raw.info = crate::decode::heap_decoder::XLOG_HEAP_INPLACE; b.stash_raw(1, raw).await.unwrap(); - inject_ordinary(&mut b, rfn, rel); - let mut drain = b.drain_committed(1, 42, 0x2000, &[], false).await.unwrap(); + let stash = inject_ordinary(rfn, rel); + let mut drain = b + .drain_committed(stash, 42, 0x2000, &[], false) + .await + .unwrap(); let err = drain .next_batch(8, usize::MAX, None) .await @@ -5057,8 +5085,11 @@ mod tests { b.stash_raw(1, multi_insert_raw(99, 100, 16417, &[5])) .await .unwrap(); - inject_ordinary(&mut b, rfn, rel); - let mut drain = b.drain_committed(1, 42, 0x2000, &[], false).await.unwrap(); + let stash = inject_ordinary(rfn, rel); + let mut drain = b + .drain_committed(stash, 42, 0x2000, &[], false) + .await + .unwrap(); let err = drain .next_batch(8, usize::MAX, None) .await @@ -5088,8 +5119,11 @@ mod tests { b.stash_raw(1, multi_insert_raw(1, 100, 16413, &[7])) .await .unwrap(); - inject_ordinary(&mut b, rfn, rel); - let mut drain = b.drain_committed(1, 42, 0x2000, &[], false).await.unwrap(); + let stash = inject_ordinary(rfn, rel); + let mut drain = b + .drain_committed(stash, 42, 0x2000, &[], false) + .await + .unwrap(); let batch = drain .next_batch(8, usize::MAX, None) .await @@ -5147,8 +5181,11 @@ mod tests { }], }; b.stash_raw(1, raw).await.unwrap(); - inject_ordinary(&mut b, rfn, rel); - let mut drain = b.drain_committed(1, 42, 0x2000, &[], false).await.unwrap(); + let stash = inject_ordinary(rfn, rel); + let mut drain = b + .drain_committed(stash, 42, 0x2000, &[], false) + .await + .unwrap(); let err = drain .next_batch(8, usize::MAX, None) .await @@ -5178,8 +5215,11 @@ mod tests { .await .unwrap(); b.stash_raw(1, update_raw(1, 150, 16415, 9)).await.unwrap(); - inject_ordinary(&mut b, rfn, rel); - let mut drain = b.drain_committed(1, 42, 0x2000, &[], false).await.unwrap(); + let stash = inject_ordinary(rfn, rel); + let mut drain = b + .drain_committed(stash, 42, 0x2000, &[], false) + .await + .unwrap(); let batch = drain .next_batch(8, usize::MAX, None) .await @@ -5229,8 +5269,7 @@ mod tests { b.stash_raw(1, multi_insert_raw(1, 300, 16417, &[2])) .await .unwrap(); - inject_ordinary_pending( - &mut b, + let stash = inject_ordinary_pending( rfn, Arc::new(commit_shape), vec![PendingSlot { @@ -5239,7 +5278,10 @@ mod tests { desc: boundary_shape, }], ); - let mut drain = b.drain_committed(1, 42, 0x2000, &[], false).await.unwrap(); + let mut drain = b + .drain_committed(stash, 42, 0x2000, &[], false) + .await + .unwrap(); let batch = drain .next_batch(8, usize::MAX, None) .await @@ -5277,8 +5319,7 @@ mod tests { b.stash_raw(1, multi_insert_raw(1, 100, 16418, &[7])) .await .unwrap(); - inject_ordinary_pending( - &mut b, + let stash = inject_ordinary_pending( rfn, owner.clone(), vec![PendingSlot { @@ -5287,7 +5328,10 @@ mod tests { desc: Arc::new(transient), }], ); - let mut drain = b.drain_committed(1, 42, 0x2000, &[], false).await.unwrap(); + let mut drain = b + .drain_committed(stash, 42, 0x2000, &[], false) + .await + .unwrap(); let batch = drain .next_batch(8, usize::MAX, None) .await @@ -5323,9 +5367,9 @@ mod tests { .collect(); assert!(spilled.contains(&(8, true)), "{spilled:?}"); assert!(spilled.contains(&(9, false)), "{spilled:?}"); - inject_ordinary(&mut b, rfn, rel); + let stash = inject_ordinary(rfn, rel); let mut drain = b - .drain_committed(1, 42, 0x2000, &[8, 9], false) + .drain_committed(stash, 42, 0x2000, &[8, 9], false) .await .unwrap(); let mut heaps = Vec::new(); @@ -5367,8 +5411,11 @@ mod tests { b.on_schema_event(2, 150, dropped_event(8)); assert!(b.stats().spill_xacts_active >= 1, "xid 1 must spill"); - let mut drain = b.drain_committed(1, 42, 0x2000, &[2], false).await.unwrap(); - assert!(drain.had_states); + let mut drain = b + .drain_committed(StashResolved::nothing_stashed(1), 42, 0x2000, &[2], false) + .await + .unwrap(); + assert!(drain.had_states()); let mut order: Vec = Vec::new(); let mut finals = Vec::new(); while let Some(batch) = drain.next_batch(2, usize::MAX, None).await.unwrap() { @@ -5426,7 +5473,10 @@ mod tests { .unwrap(); b.on_heap(heap_with_value(9, 300, 16)).await.unwrap(); - let mut drain = b.drain_committed(9, 0, 0x1000, &[], true).await.unwrap(); + let mut drain = b + .drain_committed(StashResolved::nothing_stashed(9), 0, 0x1000, &[], true) + .await + .unwrap(); let b1 = drain .next_batch(1, usize::MAX, None) .await @@ -5485,7 +5535,10 @@ mod tests { b.on_toast_chunk(chunk(57, 0, 140, b"ef"), 9).await.unwrap(); b.on_heap(heap_with_value(9, 150, 16)).await.unwrap(); - let mut drain = b.drain_committed(9, 0, 0x1000, &[], true).await.unwrap(); + let mut drain = b + .drain_committed(StashResolved::nothing_stashed(9), 0, 0x1000, &[], true) + .await + .unwrap(); let batch = drain .next_batch(usize::MAX, usize::MAX, None) .await @@ -5519,7 +5572,10 @@ mod tests { b.on_heap(heap_with_value(3, 100 + i, 512)).await.unwrap(); } assert_eq!(spill_files(tmp.path()).len(), 1); - let mut drain = b.drain_committed(3, 0, 0x5000, &[], false).await.unwrap(); + let mut drain = b + .drain_committed(StashResolved::nothing_stashed(3), 0, 0x5000, &[], false) + .await + .unwrap(); assert_eq!( b.stats().spill_bytes_active, 0, @@ -5550,14 +5606,23 @@ mod tests { b.on_heap(heap_with_value(4, 100 + i, 512)).await.unwrap(); } let total = n * 512; - let mut drain = b.drain_committed(4, 0, 0x9000, &[], false).await.unwrap(); + let mut drain = b + .drain_committed(StashResolved::nothing_stashed(4), 0, 0x9000, &[], false) + .await + .unwrap(); let read_buffers = drain .merged .as_ref() .unwrap() .sources .iter() - .filter_map(|s| s.reader.as_ref()) + .filter_map(|s| { + if let SpillSide::Reading(r) = &s.spill { + Some(r) + } else { + None + } + }) .map(|r| r.buffered_bytes() as u64) .sum::(); let mut rows = 0usize; @@ -5602,7 +5667,10 @@ mod tests { b.on_heap(heap_with_value(9, lsn + 1, 16)).await.unwrap(); } let total = u64::from(n) * (512 + CHUNK_REF_META as u64); - let mut drain = b.drain_committed(9, 0, 0x9000, &[], true).await.unwrap(); + let mut drain = b + .drain_committed(StashResolved::nothing_stashed(9), 0, 0x9000, &[], true) + .await + .unwrap(); let mut held: Vec = Vec::new(); while let Some(batch) = drain.next_batch(2, usize::MAX, None).await.unwrap() { let is_final = batch.is_final; @@ -5663,7 +5731,10 @@ mod tests { .unwrap(); b.on_heap(heap_with_value(9, lsn + 1, 16)).await.unwrap(); } - let mut drain = b.drain_committed(9, 0, 0x9000, &[], true).await.unwrap(); + let mut drain = b + .drain_committed(StashResolved::nothing_stashed(9), 0, 0x9000, &[], true) + .await + .unwrap(); let mut held: Vec = Vec::new(); while let Some(batch) = drain.next_batch(2, usize::MAX, None).await.unwrap() { let is_final = batch.is_final; @@ -5749,7 +5820,10 @@ mod tests { b.on_toast_chunk(chunk(50, 0, 100, b"aa"), 9).await.unwrap(); b.on_toast_chunk(chunk(51, 0, 102, b"bb"), 9).await.unwrap(); b.on_heap(heap_with_value(9, 110, 16)).await.unwrap(); - let mut drain = b.drain_committed(9, 0, 0x1000, &[], true).await.unwrap(); + let mut drain = b + .drain_committed(StashResolved::nothing_stashed(9), 0, 0x1000, &[], true) + .await + .unwrap(); let err = drain .next_batch(usize::MAX, usize::MAX, None) .await @@ -5765,8 +5839,11 @@ mod tests { async fn drain_committed_unknown_xid_yields_no_batches() { let tmp = tempdir().unwrap(); let mut b = XactBuffer::new(cfg(tmp.path().to_path_buf())).unwrap(); - let mut drain = b.drain_committed(999, 0, 0x100, &[], false).await.unwrap(); - assert!(!drain.had_states); + let mut drain = b + .drain_committed(StashResolved::nothing_stashed(999), 0, 0x100, &[], false) + .await + .unwrap(); + assert!(!drain.had_states()); assert!(drain.next_batch(10, 1000, None).await.unwrap().is_none()); drain.finish().await.unwrap(); assert_eq!(b.stats().commits_unknown_xid, 1); diff --git a/tests/backfill_staging_e2e.rs b/tests/backfill_staging_e2e.rs index cb2e2b3f..5b30ee95 100644 --- a/tests/backfill_staging_e2e.rs +++ b/tests/backfill_staging_e2e.rs @@ -13,7 +13,7 @@ use walshadow::backfill_staging::{self, StagingRel, StagingSession}; use walshadow::backfill_types::BackupRequest; use walshadow::ch_emitter::{EmitterConfig, EmitterStats}; use walshadow::copy_backfill::CopyBackfiller; -use walshadow::desc_log::{DescLogIdentity, DescriptorLog}; +use walshadow::desc_log::{BatchRecord, DescLogIdentity, DescriptorLog, LogEntry, LogValue}; use walshadow::mapping::{ColumnMapping, MappingHandle, TableMapping, TableTarget, mapping_handle}; use walshadow::runtime_config::InitialLoadMode; use walshadow::schema::{RelDescriptor, RelName}; @@ -60,7 +60,7 @@ impl Fixture { DescriptorLog::open( &tmp.path().join("descriptors"), DescLogIdentity { - pg_major: bridge.info().unwrap().pg_version_num / 10000, + pg_major: bridge.info().pg_version_num / 10000, system_id: source .psql_one("SELECT system_identifier FROM pg_control_system()") .unwrap(), @@ -299,6 +299,7 @@ async fn staged_backfill_resumes_each_publish_phase() { let dir = tempfile::tempdir().unwrap(); fx.ch.query("TRUNCATE TABLE default.t").unwrap(); fx.ch.query("INSERT INTO default.t (id, name, _lsn, _is_deleted) VALUES (9, 'stale', 99, false), (8, 'boundary', 100, false), (2, 'live', 101, false), (3, 'deleted', 102, true)").unwrap(); + write_held_pending(&fx, dir.path()); let rel = fx.prepare_staging().await; fx.ch.query("INSERT INTO default.t__wsstg (id, name, _lsn) VALUES (1, 'snapshot', 100), (2, 'old', 100), (3, 'old', 100)").unwrap(); let uuid = session @@ -308,7 +309,9 @@ async fn staged_backfill_resumes_each_publish_phase() { .unwrap(); write_swapped_ledger(dir.path(), &uuid); if phase != "before_exchange" { - session.exchange(&rel).await.unwrap(); + fx.ch + .query("EXCHANGE TABLES default.t AND default.t__wsstg") + .unwrap(); assert_eq!( session.table_uuid("default", "t").await.unwrap().as_deref(), Some(uuid.as_str()) @@ -331,14 +334,21 @@ async fn staged_backfill_resumes_each_publish_phase() { .note_opt_in(&fx.desc, InitialLoadMode::Copy, 999) .await; wait_done(&backfiller, dir.path()).await; - assert_eq!(fx.rows(), "1\tsnapshot\n2\tlive", "{phase}"); + assert_eq!(fx.rows(), "1\tsnapshot\n2\tlive\n5\tpending", "{phase}"); assert_eq!( fx.ch .query("SELECT id, max(_lsn), argMax(_is_deleted, _lsn) FROM default.t GROUP BY id ORDER BY id") .unwrap(), - "1\t100\tfalse\n2\t101\tfalse\n3\t102\ttrue", + "1\t100\tfalse\n2\t101\tfalse\n3\t102\ttrue\n5\t100\tfalse", "{phase}" ); + assert_eq!(fx.ch.query("EXISTS default.t__wspending").unwrap(), "0"); + assert!( + PendingLedger::load(dir.path(), fx.system_id()) + .await + .unwrap() + .is_empty() + ); assert_eq!( session.table_uuid("default", "t__wsstg").await.unwrap(), None @@ -359,7 +369,7 @@ async fn staged_backfill_resumes_each_publish_phase() { .note_opt_in(&fx.desc, InitialLoadMode::BaseBackup, 1000) .await; assert_eq!(restarted.pending_count(), 0); - assert_eq!(fx.rows(), "1\tsnapshot\n2\tlive", "{phase}"); + assert_eq!(fx.rows(), "1\tsnapshot\n2\tlive\n5\tpending", "{phase}"); } } @@ -367,6 +377,47 @@ fn read_ledger(dir: &Path) -> toml::Value { toml::from_str(&std::fs::read_to_string(dir.join("backfills.toml")).unwrap()).unwrap() } +/// Pass held its pending rows before EXCHANGE; xid 500 committed meanwhile +fn write_held_pending(fx: &Fixture, dir: &Path) { + fx.ch + .query("CREATE TABLE default.t__wspending AS default.t ENGINE = MergeTree ORDER BY tuple() PRIMARY KEY tuple() PARTITION BY tuple()") + .unwrap(); + fx.ch + .query("ALTER TABLE default.t__wspending ADD COLUMN _ws_xmin UInt32, ADD COLUMN _ws_xmax UInt32, ADD COLUMN _ws_infomask UInt16") + .unwrap(); + fx.ch + .query("INSERT INTO default.t__wspending (id, name, _lsn, _ws_xmin) VALUES (5, 'pending', 100, 500)") + .unwrap(); + std::fs::write( + walshadow::visibility_pending::ledger_path(dir), + format!( + "version = 1\nsystem_id = {}\n[[carry]]\nnamespace = 'public'\nrelname = 't'\n\ + database = 'default'\ntable = 't'\nstart_lsn = '0/64'\ncommitted = [500]\n\ + held = true\n", + fx.system_id() + ), + ) + .unwrap(); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn boot_discards_held_pending_without_swap() { + if !fx::tools::requirements_available() { + return; + } + let fx = Fixture::new().await; + let dir = tempfile::tempdir().unwrap(); + // Crash after hold, before swap mark, then opt-out drops the backfill entry + write_held_pending(&fx, dir.path()); + fx.backfiller(dir.path()).await; + assert!( + PendingLedger::load(dir.path(), fx.system_id()) + .await + .unwrap() + .is_empty() + ); +} + fn write_swapped_ledger(dir: &Path, uuid: &str) { std::fs::write(dir.join("backfills.toml"), format!("version = 1\n[[backfill]]\nnamespace = 'public'\nrelname = 't'\ns_lsn = '0/64'\ndone = false\nmode = 'base_backup'\nswapped = true\nstaging_uuid = '{uuid}'\n")).unwrap(); } @@ -397,6 +448,7 @@ async fn staged_schema_change_discards_load_and_keeps_retry_pending() { .unwrap() .unwrap(); write_swapped_ledger(dir.path(), &uuid); + write_held_pending(&fx, dir.path()); fx.ch .query("ALTER TABLE default.t ADD COLUMN extra String DEFAULT 'new'") .unwrap(); @@ -417,6 +469,13 @@ async fn staged_schema_change_discards_load_and_keeps_retry_pending() { ); assert_eq!(backfiller.pending_by_mode(), [0, 1, 0]); assert_eq!(fx.rows(), "9\tlive"); + assert!( + PendingLedger::load(dir.path(), fx.system_id()) + .await + .unwrap() + .is_empty(), + "discarded load never promotes its pending rows" + ); assert_eq!( session.table_uuid("default", "t__wsstg").await.unwrap(), None @@ -635,3 +694,95 @@ async fn failed_backup_can_disable_copy_fallback() { ); assert_eq!(fx.rows(), ""); } + +/// Archive every WAL file in source's `pg_wal`, current segment included, so +/// the gap through the opt-in boundary is fetchable +async fn archive_wal( + source: &Shadow, + settings: &walrus::config::Settings, + storage: &walrus::storage::DynStorage, +) { + let pg_wal = source.config().data_dir.join("pg_wal"); + for entry in std::fs::read_dir(pg_wal).unwrap() { + let path = entry.unwrap().path(); + let name = path.file_name().unwrap().to_str().unwrap(); + if name.len() == 24 && name.chars().all(|c| c.is_ascii_hexdigit()) { + walrus::pg::wal::push::handle(settings, storage.clone(), &path) + .await + .unwrap(); + } + } +} + +/// Backup predates the opt-in: walk loads the backup image, gap replay +/// carries changes between backup redo and the boundary +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn object_store_opt_in_replays_archive_gap() { + if !fx::tools::requirements_available() { + return; + } + let mut fx = Fixture::new().await; + // Live capture fills the log in production; replay decodes through it + fx.log + .seed( + BatchRecord { + captured_at: 0x1000, + commit_lsn: 0, + observations: Vec::new(), + ambiguities: Vec::new(), + entries: vec![Arc::new(LogEntry { + valid_from: 0x1000, + oid: fx.desc.oid, + rfn: fx.desc.rfn, + value: LogValue::Present(fx.desc.clone()), + })], + }, + 0x1000, + ) + .await + .unwrap(); + let root = tempfile::tempdir().unwrap(); + let storage: walrus::storage::DynStorage = + Arc::new(walrus::storage::fs::FsStorage::new(root.path()).unwrap()); + let settings = walrus::config::Settings { + storage: walrus::config::StorageSettings::Fs { + path: root.path().to_string_lossy().into_owned(), + }, + compression: walrus::compression::Method::None, + compression_level: 0, + ..Default::default() + }; + walrus::pg::backup::push::handle( + &settings, + storage.clone(), + walrus::pg::backup::push::PushArgs::default(), + fx::pg_cfg(&fx.source, "backfill-push"), + ) + .await + .unwrap(); + fx.source + .psql_one( + "UPDATE public.t SET name = 'updated' WHERE id = 1; \ + DELETE FROM public.t WHERE id = 2; \ + INSERT INTO public.t VALUES (4, 'four')", + ) + .unwrap(); + let s_lsn = fx.source.psql_one("SELECT pg_current_wal_lsn()").unwrap(); + let s_lsn = walrus::pg::backup::parse_pg_lsn(&s_lsn).unwrap(); + fx.source.psql_one("SELECT pg_switch_wal()").unwrap(); + archive_wal(&fx.source, &settings, &storage).await; + fx.emitter.backup = Some(settings); + + let dir = tempfile::tempdir().unwrap(); + let backfiller = fx.backfiller(dir.path()).await; + backfiller + .note_opt_in(&fx.desc, InitialLoadMode::ObjectStore, s_lsn) + .await; + wait_done(&backfiller, dir.path()).await; + assert_eq!(fx.rows(), "1\tupdated\n3\tthree\n4\tfour"); + assert_eq!(fx.stats.backfill_copy_rows.load(Ordering::Relaxed), 0); + assert_eq!( + read_ledger(dir.path())["backfill"][0]["mode"].as_str(), + Some("object_store") + ); +} diff --git a/tests/bootstrap_pipeline_ch.rs b/tests/bootstrap_pipeline_ch.rs index 5a03c651..ddf74f05 100644 --- a/tests/bootstrap_pipeline_ch.rs +++ b/tests/bootstrap_pipeline_ch.rs @@ -209,9 +209,7 @@ async fn bootstrap_tail_fans_out_n2() { "contiguous-done watermark at start_lsn", ); - drop(msg_tx); - drop(ack); - tail.join().await; + tail.close(msg_tx, ack).await; assert!(fatal.message().is_none(), "no fatal: {:?}", fatal.message()); // Every fed row landed; out-of-order batch completion across the two diff --git a/tests/bridge.rs b/tests/bridge.rs index 8842d61a..29688dc0 100644 --- a/tests/bridge.rs +++ b/tests/bridge.rs @@ -26,7 +26,7 @@ use walshadow::oracle::{Oracle, OracleCell, OracleColumnBuf, OracleRequestColumn use walshadow::pg::socket_conninfo; use walshadow::schema::{NUMERICOID, ReplIdent}; use walshadow::shadow::{BridgeConf, Shadow, ShadowConfig}; -use walshadow::shadow_catalog::{CatalogError, ShadowCatalog, ShadowCatalogConfig}; +use walshadow::shadow_catalog::{CatalogError, ParkedAt, ShadowCatalog, ShadowCatalogConfig}; /// Build tree holding `walshadow.so`, fed to shadow as `dynamic_library_path`. /// Module is not optional, so an unbuilt tree fails rather than skips @@ -239,7 +239,7 @@ async fn bridge_hello_and_encode_native() { let guard = start_pg(&tmp, ports::PG_SHADOW_PORT); let bridge = dial(&guard.sh).await; - let info = bridge.info().expect("hello"); + let info = bridge.info(); assert_eq!(info.proto, PROTO_VERSION); assert_eq!(info.projection, PROJECTION_VERSION); assert!(info.pg_version_num >= 160000, "{info:?}"); @@ -857,7 +857,7 @@ async fn bridge_overlay_descriptors_track_open_ddl() { idle.batch_execute("BEGIN").await.expect("begin idle"); let idle_xid = top_xid(&idle).await; let overlay = cat - .fetch_overlay_descriptors(&[oid_t], idle_xid, 0) + .fetch_overlay_descriptors(&[oid_t], idle_xid, &ParkedAt::assume_for_test(0)) .await .expect("overlay batch"); assert_eq!( @@ -882,7 +882,7 @@ async fn bridge_overlay_descriptors_track_open_ddl() { let oid_u = oid_of(&ddl, "fresh.u").await; let mut descs = cat - .fetch_overlay_descriptors(&[oid_t, oid_u], xid, 0) + .fetch_overlay_descriptors(&[oid_t, oid_u], xid, &ParkedAt::assume_for_test(0)) .await .expect("overlay under open ddl"); descs.sort_by_key(|d| d.oid); @@ -930,7 +930,7 @@ async fn bridge_overlay_descriptors_track_open_ddl() { // Replay off the asserted boundary is the caller's whole basis for reading // uncommitted rows, so it fails rather than answering let err = cat - .fetch_overlay_descriptors(&[oid_t], xid, 0x1000) + .fetch_overlay_descriptors(&[oid_t], xid, &ParkedAt::assume_for_test(0x1000)) .await .unwrap_err(); assert!( @@ -943,7 +943,7 @@ async fn bridge_overlay_descriptors_track_open_ddl() { ddl.batch_execute("ROLLBACK").await.expect("undo"); let after = cat - .fetch_overlay_descriptors(&[oid_t, oid_u], xid, 0) + .fetch_overlay_descriptors(&[oid_t, oid_u], xid, &ParkedAt::assume_for_test(0)) .await .expect("overlay after rollback"); assert_eq!(after, committed, "aborted tree still visible"); @@ -1258,7 +1258,7 @@ async fn bridge_committed_read_falls_back_when_replay_moves() { // The caller holds the boundary an overlay read is about, so nothing else // can serve that question let err = cat - .fetch_overlay_descriptors(&[oid_t], 700, 0x1000) + .fetch_overlay_descriptors(&[oid_t], 700, &ParkedAt::assume_for_test(0x1000)) .await .unwrap_err(); assert!( @@ -1533,7 +1533,7 @@ async fn bridge_worker_pool_serves_concurrent_requests() { assert_eq!(pooled.pool_size(), WORKERS, "one socket per worker"); assert_eq!(single.pool_size(), 1); // Every slot dialled its own HELLO; a mismatch would have failed connect - assert!(pooled.info().is_some()); + assert_eq!(pooled.info(), single.info()); // Wide enough that requests are still overlapping when the last one is // dispatched, which is the state a shared slot would have to serialize diff --git a/tests/common/inproc_harness.rs b/tests/common/inproc_harness.rs index b5949c29..6e333e49 100644 --- a/tests/common/inproc_harness.rs +++ b/tests/common/inproc_harness.rs @@ -53,7 +53,7 @@ use walshadow::record::{ use walshadow::schema::RelName; use walshadow::segment_sink::DirSegmentSink; use walshadow::shadow::{Shadow, ShadowConfig}; -use walshadow::shadow_catalog::{ShadowCatalog, ShadowCatalogConfig}; +use walshadow::shadow_catalog::{ParkedAt, ShadowCatalog, ShadowCatalogConfig}; use walshadow::source_feed::{SourceEvent, SourceFeed, StandbyStatus}; use walshadow::wal_stream::WalStream; use walshadow::xact_buffer::{BufferingDecoderSink, SubxactTracker, XactBuffer, XactBufferConfig}; @@ -580,13 +580,15 @@ impl RecordSink for PipelineSinks { .await .map_err(|e| SinkError::Other(format!("harness boundary wait: {e}")))?; self.capture.charge_hold(info, std::time::Duration::ZERO); + // Harness pump withholds successors while this runs + let parked = ParkedAt::assume_for_test(record.next_lsn); self.capture - .capture_boundary(info, record.source_lsn, record.next_lsn) + .capture_boundary(info, record.source_lsn, &parked) .await?; } else if matches!(info.kind, BoundaryKind::Commit) { // Save an empty batch without waiting, as daemon does self.capture - .capture_boundary(info, record.source_lsn, record.next_lsn) + .cover_unparked(info, record.source_lsn, record.next_lsn) .await?; } } @@ -682,7 +684,7 @@ async fn build_pipeline_inner( .with_status_interval(Duration::from_millis(500)); let ident = feed.identify_system().await.expect("IDENTIFY_SYSTEM"); let aligned = WalStream::align_down(ident.xlogpos, WAL_SEG_SIZE); - let mut stream = WalStream::new(ident.timeline, WAL_SEG_SIZE, Pos::new(aligned)).unwrap(); + let mut stream = WalStream::builder(ident.timeline, WAL_SEG_SIZE, Pos::new(aligned)).unwrap(); stream.set_bytes_sink(Box::new(walshadow::shadow_stream::ShadowStreamSink::new( shadow_stream_state, ))); @@ -771,7 +773,7 @@ async fn build_pipeline_inner( .await .expect("shadow db oid"); stream.filter_mut().set_target_db(shadow_db_oid); - let smgr_markers = stream.filter_mut().smgr_markers(); + let smgr_markers = stream.filter().smgr_markers(); let desc_log = Arc::new( walshadow::desc_log::DescriptorLog::open( &spill_dir, @@ -852,11 +854,12 @@ async fn build_pipeline_inner( None, toml::Table::new(), mapping.clone(), + walshadow::config::ResolverBoot { + shadow_toast: stream.filter().shadow_rels().map(|rels| rels.held()), + ..Default::default() + }, ); - // Apply daemon's TOAST admission check to opt-ins - if let Some(rels) = stream.filter().shadow_rels() { - config_resolver.bind_shadow_toast(rels.held()); - } + let stream = stream.start(); let ddl_cfg = DdlConfig::from_resolved( &config_rx.borrow(), emitter_cfg.database.clone(), @@ -953,20 +956,18 @@ async fn build_pipeline_inner( resume_floor: resume_floor.clone(), budget: None, }; - let (mut reorder, handle) = pcfg + let (booting, handle) = pcfg .spawn(emitter_ack.clone()) .await .expect("spawn decode+insert pipeline"); - // Prior run's drop segment is below the boot head; retire its queued - // mirrors now — no commit will replay the drop (mirrors bin/stream/session.rs) - reorder - .flush_due_retires() - .await - .expect("boot flush of due toast-mirror retires"); - reorder - .apply_boot_events(desc_log.active_present_at(ident.xlogpos), ident.xlogpos) + let reorder = booting + .boot( + None, + desc_log.active_present_at(ident.xlogpos), + ident.xlogpos, + ) .await - .expect("boot Added pass over descriptor log"); + .expect("boot reorder sink"); let mut decoder = BufferingDecoderSink::new( walshadow::desc_log::DescriptorLogs::single(desc_log.clone()), diff --git a/tests/emitter_budget_flush.rs b/tests/emitter_budget_flush.rs index 26f2055d..cae4cbf5 100644 --- a/tests/emitter_budget_flush.rs +++ b/tests/emitter_budget_flush.rs @@ -26,6 +26,7 @@ use walshadow::ch::CompressionChoice; use walshadow::ch_emitter::{EmitterConfig, EmitterStats}; use walshadow::heap_decoder::{ColumnValue, CommittedTuple, DecodedHeap, DecodedTuple, HeapOp}; use walshadow::mapping::{ColumnMapping, TableMapping, TableTarget}; +use walshadow::pipeline::ack::Publish; use walshadow::pipeline::batcher::{BatcherMsg, RoutedRow}; use walshadow::pipeline::{Fatal, tail}; use walshadow::pos::Pos; @@ -176,11 +177,11 @@ async fn budget_trips_seal_complete_inserts() { ); const N: i32 = 5; let commit_lsn = 0xC0FFEE; - ack.register(0, commit_lsn); + let seq = ack.seqs(0).open(commit_lsn, Publish::Commit); for i in 0..N { msg_tx .send(BatcherMsg::Row(RoutedRow { - seq: 0, + seq: seq.seq(), rel: rel.clone(), route: route.clone(), committed: tuple(i, 0x1000 + i as u64, commit_lsn), @@ -189,7 +190,7 @@ async fn budget_trips_seal_complete_inserts() { .await .expect("route row"); } - ack.placed(0, N as u64); + seq.place(N as u64); let (reply_tx, reply_rx) = tokio::sync::oneshot::channel(); msg_tx .send(BatcherMsg::FlushAll(reply_tx)) @@ -197,9 +198,7 @@ async fn budget_trips_seal_complete_inserts() { .expect("send flush"); reply_rx.await.expect("flush ack"); ack.wait_through(1).await.expect("ack collector alive"); - drop(msg_tx); - drop(ack); - tail_parts.join().await; + tail_parts.close(msg_tx, ack).await; assert!(fatal.message().is_none(), "no fatal: {:?}", fatal.message()); // The contiguous-done watermark reaches the seq's commit lsn. diff --git a/tests/emitter_native_types.rs b/tests/emitter_native_types.rs index 854f0074..1dae462c 100644 --- a/tests/emitter_native_types.rs +++ b/tests/emitter_native_types.rs @@ -23,6 +23,7 @@ use walshadow::ch_emitter::{EmitterConfig, EmitterStats}; use walshadow::codecs::NumericKind; use walshadow::heap_decoder::{ColumnValue, CommittedTuple, DecodedHeap, DecodedTuple, HeapOp}; use walshadow::mapping::{ColumnMapping, TableMapping, TableTarget}; +use walshadow::pipeline::ack::Publish; use walshadow::pipeline::batcher::{BatcherMsg, RoutedRow}; use walshadow::pipeline::{Fatal, tail}; use walshadow::pos::{EmitterAck, Monotone}; @@ -168,10 +169,10 @@ async fn native_numeric_time_timetz_round_trip() { // Drive the single row through the tail: register the seq, route the // row, seal with FlushAll, wait for it durable, then drain. - ack.register(0, tuple.commit_lsn); + let seq = ack.seqs(0).open(tuple.commit_lsn, Publish::Commit); msg_tx .send(BatcherMsg::Row(RoutedRow { - seq: 0, + seq: seq.seq(), rel, route: walshadow::emit::route::RouteSnapshot::freeze( mapping, @@ -183,7 +184,7 @@ async fn native_numeric_time_timetz_round_trip() { })) .await .expect("route row"); - ack.placed(0, 1); + seq.place(1); let (reply_tx, reply_rx) = tokio::sync::oneshot::channel(); msg_tx .send(BatcherMsg::FlushAll(reply_tx)) @@ -191,9 +192,7 @@ async fn native_numeric_time_timetz_round_trip() { .expect("send flush"); reply_rx.await.expect("flush ack"); ack.wait_through(1).await.expect("ack collector alive"); - drop(msg_tx); - drop(ack); - tail_parts.join().await; + tail_parts.close(msg_tx, ack).await; assert!(fatal.message().is_none(), "no fatal: {:?}", fatal.message()); let row = ch diff --git a/tests/emitter_tls.rs b/tests/emitter_tls.rs index 6ad47545..316273d0 100644 --- a/tests/emitter_tls.rs +++ b/tests/emitter_tls.rs @@ -39,6 +39,7 @@ use walshadow::ch::CompressionChoice; use walshadow::ch_emitter::{EmitterConfig, EmitterStats}; use walshadow::heap_decoder::{ColumnValue, CommittedTuple, DecodedHeap, DecodedTuple, HeapOp}; use walshadow::mapping::{ColumnMapping, TableMapping, TableTarget}; +use walshadow::pipeline::ack::Publish; use walshadow::pipeline::batcher::{BatcherMsg, RoutedRow}; use walshadow::pipeline::{Fatal, tail}; use walshadow::pos::{EmitterAck, Monotone}; @@ -369,10 +370,10 @@ async fn emitter_tls_round_trip() { commit_ts: 1_000_000, commit_lsn: 0xABCD, }; - ack.register(0, tuple.commit_lsn); + let seq = ack.seqs(0).open(tuple.commit_lsn, Publish::Commit); msg_tx .send(BatcherMsg::Row(RoutedRow { - seq: 0, + seq: seq.seq(), rel, route: walshadow::emit::route::RouteSnapshot::freeze( mapping, @@ -384,7 +385,7 @@ async fn emitter_tls_round_trip() { })) .await .expect("route row"); - ack.placed(0, 1); + seq.place(1); let (reply_tx, reply_rx) = tokio::sync::oneshot::channel(); msg_tx .send(BatcherMsg::FlushAll(reply_tx)) @@ -392,9 +393,7 @@ async fn emitter_tls_round_trip() { .expect("send flush"); reply_rx.await.expect("flush ack"); ack.wait_through(1).await.expect("ack collector alive"); - drop(msg_tx); - drop(ack); - tail_parts.join().await; + tail_parts.close(msg_tx, ack).await; assert!(fatal.message().is_none(), "no fatal: {:?}", fatal.message()); let row = ch diff --git a/tests/fpi_user_pages.rs b/tests/fpi_user_pages.rs index 06ac3854..f5233476 100644 --- a/tests/fpi_user_pages.rs +++ b/tests/fpi_user_pages.rs @@ -156,7 +156,7 @@ async fn attach(source: &Shadow) -> (SourceFeed, WalStream) { .with_status_interval(Duration::from_millis(500)); let ident = feed.identify_system().await.expect("IDENTIFY_SYSTEM"); let aligned = WalStream::align_down(ident.xlogpos, WAL_SEG_SIZE); - let mut stream = WalStream::new(ident.timeline, WAL_SEG_SIZE, Pos::new(aligned)).unwrap(); + let mut stream = WalStream::builder(ident.timeline, WAL_SEG_SIZE, Pos::new(aligned)).unwrap(); stream.filter_mut().set_target_db(current_db_oid(source)); { let sql_client = feed.sql_client().await.expect("sql client"); @@ -175,7 +175,7 @@ async fn attach(source: &Shadow) -> (SourceFeed, WalStream) { feed.start_physical_replication(None, aligned, ident.timeline) .await .expect("START_REPLICATION"); - (feed, stream) + (feed, stream.start()) } #[tokio::test(flavor = "multi_thread", worker_threads = 2)] diff --git a/tests/multi_segment_filter.rs b/tests/multi_segment_filter.rs index bb2cd692..249b9a44 100644 --- a/tests/multi_segment_filter.rs +++ b/tests/multi_segment_filter.rs @@ -309,7 +309,6 @@ impl RecordSink for SharedCollectingSink { next_lsn: r.next_lsn, page_magic: r.page_magic, route: r.route, - catalog_boundary: r.catalog_boundary, boundary_info: r.boundary_info.clone(), aborted_tree: r.aborted_tree.clone(), defer_catalog_decode: r.defer_catalog_decode, diff --git a/tests/multixact_visibility.rs b/tests/multixact_visibility.rs index db41f1ee..6e07c7a2 100644 --- a/tests/multixact_visibility.rs +++ b/tests/multixact_visibility.rs @@ -117,7 +117,7 @@ fn multixact_updater_matches_live_pg() { // its delete may predate WAL coverage — must gate, not resurrect let mut xact = PgXactAccum::new(); xact.insert_segment(0, std::fs::read(data_dir.join("pg_xact/0000")).unwrap()); - let patch = PgXactPatch::new(); + let patch = PgXactPatch::new().seal(); let view = PgXactView::new(&xact, &patch).with_multixact(&multi); assert_eq!( tuple_visibility( diff --git a/tests/oracle_types_e2e.rs b/tests/oracle_types_e2e.rs index e4898af6..d6c59ffe 100644 --- a/tests/oracle_types_e2e.rs +++ b/tests/oracle_types_e2e.rs @@ -94,7 +94,7 @@ async fn run_oracle_stats( .await .expect("bridge connect"); assert!( - bridge.info().expect("hello").in_recovery, + bridge.info().in_recovery, "shadow must serve decode while in recovery", ); let oracle = Arc::new(Oracle::new(Arc::new(bridge))); diff --git a/tests/resume_shadow_relations.rs b/tests/resume_shadow_relations.rs index b37b903c..b4ee215c 100644 --- a/tests/resume_shadow_relations.rs +++ b/tests/resume_shadow_relations.rs @@ -123,7 +123,7 @@ async fn retained_routes_prevent_invalid_page_recovery_after_resume() { }; let bytes = fs::read(raw.join(seg.format())).unwrap(); for (sh, durable) in [(&bad, false), (&good, true)] { - let mut stream = WalStream::new(1, WAL_SEG_SIZE, Pos::new(resume)).unwrap(); + let mut stream = WalStream::builder(1, WAL_SEG_SIZE, Pos::new(resume)).unwrap(); if durable { stream .filter_mut() @@ -143,6 +143,7 @@ async fn retained_routes_prevent_invalid_page_recovery_after_resume() { .unwrap(); let mut records = CollectingRecordSink::default(); let mut sink = DirSegmentSink::new(sh.config().filter_out_dir.clone()).unwrap(); + let mut stream = stream.start(); stream .push(resume, &bytes, &mut records, &mut sink) .await diff --git a/tests/toast_mode_e2e.rs b/tests/toast_mode_e2e.rs index 5911cf0f..2f1b403e 100644 --- a/tests/toast_mode_e2e.rs +++ b/tests/toast_mode_e2e.rs @@ -171,7 +171,7 @@ async fn unchanged_toast_pointer_resolves_out_of_shadow() { assert!( pipeline .stream - .filter_mut() + .filter() .shadow_rels() .is_some_and(|rels| rels.contains(&(db_oid, toast_filenode))), "shadow mode must route the TOAST relation's records to shadow", @@ -288,7 +288,7 @@ async fn unchanged_toast_pointer_fills_null_without_store() { .await; assert!( - pipeline.stream.filter_mut().shadow_rels().is_none(), + pipeline.stream.filter().shadow_rels().is_none(), "disabled mode must not replay user heaps on shadow", ); diff --git a/tests/toast_rewrite_e2e.rs b/tests/toast_rewrite_e2e.rs index c2a68413..3cbef2e6 100644 --- a/tests/toast_rewrite_e2e.rs +++ b/tests/toast_rewrite_e2e.rs @@ -592,7 +592,9 @@ async fn alter_rewrite_link_swap_retires_old_mirror() { assert_eq!(stats.toast_mirror_retires.load(Ordering::Relaxed), 0); pipeline .resume_floor - .join(walshadow::pos::Pos::new(u64::MAX)); + .publish(walshadow::pos::Durable::assume_for_test( + walshadow::pos::Pos::new(u64::MAX), + )); let driver = fx::spawn_workload( &source, vec![ diff --git a/tests/toast_tombstone_e2e.rs b/tests/toast_tombstone_e2e.rs index bec59480..53c917ed 100644 --- a/tests/toast_tombstone_e2e.rs +++ b/tests/toast_tombstone_e2e.rs @@ -418,7 +418,9 @@ async fn tombstones_supersede_then_truncate_wipes_then_drop_retires() { // stand-in); the next commit executes the queued retire. pipeline .resume_floor - .join(walshadow::pos::Pos::new(u64::MAX)); + .publish(walshadow::pos::Durable::assume_for_test( + walshadow::pos::Pos::new(u64::MAX), + )); let driver = fx::spawn_workload( &source, vec![ diff --git a/tests/toast_truncate_drop_e2e.rs b/tests/toast_truncate_drop_e2e.rs index 55af0079..21bd8b1a 100644 --- a/tests/toast_truncate_drop_e2e.rs +++ b/tests/toast_truncate_drop_e2e.rs @@ -193,7 +193,9 @@ async fn cold_restart_drop_retires_mirror() { ); pipeline .resume_floor - .join(walshadow::pos::Pos::new(u64::MAX)); + .publish(walshadow::pos::Durable::assume_for_test( + walshadow::pos::Pos::new(u64::MAX), + )); let driver = fx::spawn_workload( &source, vec![ diff --git a/tests/vacuum_catalog_churn.rs b/tests/vacuum_catalog_churn.rs index 4831172e..db6bed71 100644 --- a/tests/vacuum_catalog_churn.rs +++ b/tests/vacuum_catalog_churn.rs @@ -24,7 +24,7 @@ use walshadow::segment_sink::DirSegmentSink; use walshadow::shadow::{Shadow, ShadowConfig}; use walshadow::shadow_stream::ShadowStreamSink; use walshadow::source_feed::{SourceEvent, SourceFeed, StandbyStatus}; -use walshadow::wal_stream::WalStream; +use walshadow::wal_stream::{WalStream, WalStreamBuilder}; fn make_source(tmp: &tempfile::TempDir, port: u16) -> Shadow { let mut cfg = ShadowConfig::new(tmp.path().join("source-data"), tmp.path().join("filtered")); @@ -360,7 +360,7 @@ async fn phase( } /// Attach replication feed and filter to source -async fn attach(source: &Shadow, app_name: &str) -> (SourceFeed, WalStream) { +async fn attach(source: &Shadow, app_name: &str) -> (SourceFeed, WalStreamBuilder) { let cfg = source.config(); let pgcfg = PgConfig { host: cfg.socket_dir.to_string_lossy().into_owned(), @@ -378,7 +378,7 @@ async fn attach(source: &Shadow, app_name: &str) -> (SourceFeed, WalStream) { .with_status_interval(Duration::from_millis(500)); let ident = feed.identify_system().await.expect("IDENTIFY_SYSTEM"); let aligned = WalStream::align_down(ident.xlogpos, WAL_SEG_SIZE); - let mut stream = WalStream::new(ident.timeline, WAL_SEG_SIZE, Pos::new(aligned)).unwrap(); + let mut stream = WalStream::builder(ident.timeline, WAL_SEG_SIZE, Pos::new(aligned)).unwrap(); stream.filter_mut().set_target_db(current_db_oid(source)); { let sql_client = feed.sql_client().await.expect("sql client"); @@ -424,7 +424,8 @@ async fn maintenance_traffic_costs_no_catalog_boundary() { .expect("seed schema"); let names = Names::load(&source); - let (mut feed, mut stream) = attach(&source, "vacuum-census").await; + let (mut feed, stream) = attach(&source, "vacuum-census").await; + let mut stream = stream.start(); let mut segs = DirSegmentSink::new(tmp.path().join("out")).expect("out dir"); let mut buf = Vec::with_capacity(64 * 1024); @@ -536,6 +537,7 @@ async fn maintenance_traffic_parks_the_pump_for_nothing() { let (mut feed, mut stream) = attach(source, "vacuum-hold").await; stream.set_bytes_sink(Box::new(ShadowStreamSink::new(shadow_state.clone()))); + let mut stream = stream.start(); let mut segs = DirSegmentSink::new(clusters.shadow_filter_dir.clone()).expect("filter dir"); let mut buf = Vec::with_capacity(64 * 1024); let mut sink = HoldingCensus { diff --git a/tests/wal_stream_chunk_boundary.rs b/tests/wal_stream_chunk_boundary.rs index 66ef0a67..e8c47fa7 100644 --- a/tests/wal_stream_chunk_boundary.rs +++ b/tests/wal_stream_chunk_boundary.rs @@ -168,7 +168,7 @@ async fn shadow_stream_sink_receives_byte_exact_wire_stream() { let bytes = load_segment(&path).await.expect("load fixture"); let seg_size = bytes.len() as u64; - let mut stream = WalStream::new(1, seg_size, Pos::ZERO).expect("stream new"); + let mut stream = WalStream::builder(1, seg_size, Pos::ZERO).expect("stream new"); let mut rec_sink = CollectingRecordSink::default(); let mut seg_sink = CollectingSegmentSink::default(); @@ -186,6 +186,7 @@ async fn shadow_stream_sink_receives_byte_exact_wire_stream() { let sink = ShadowStreamSink::new(state.clone()); stream.set_bytes_sink(Box::new(sink)); + let mut stream = stream.start(); for (i, b) in bytes.iter().enumerate() { stream .push( diff --git a/tests/wal_stream_e2e.rs b/tests/wal_stream_e2e.rs index cd9e483b..9c3a575f 100644 --- a/tests/wal_stream_e2e.rs +++ b/tests/wal_stream_e2e.rs @@ -351,7 +351,7 @@ async fn pre_rotated_pg_class_seed_keeps_catalog_writes() { .with_status_interval(Duration::from_millis(500)); let ident = feed.identify_system().await.expect("IDENTIFY_SYSTEM"); let aligned = WalStream::align_down(ident.xlogpos, WAL_SEG_SIZE); - let mut stream = WalStream::new(ident.timeline, WAL_SEG_SIZE, Pos::new(aligned)).unwrap(); + let mut stream = WalStream::builder(ident.timeline, WAL_SEG_SIZE, Pos::new(aligned)).unwrap(); // Seed *before* START_REPLICATION. Without this line the test // would catch the regression: tracker would never learn the @@ -409,6 +409,7 @@ async fn pre_rotated_pg_class_seed_keeps_catalog_writes() { let deadline = std::time::Instant::now() + Duration::from_secs(15); let mut segments_shipped = 0u64; + let mut stream = stream.start(); let mut prev = stream.dispatched_lsn(); while segments_shipped < 1 && std::time::Instant::now() < deadline { let apply_lsn = stream.dispatched_lsn(); diff --git a/tests/wal_stream_throughput.rs b/tests/wal_stream_throughput.rs index 86684c68..4a84a391 100644 --- a/tests/wal_stream_throughput.rs +++ b/tests/wal_stream_throughput.rs @@ -162,13 +162,14 @@ async fn run_case( record_sink: &mut (dyn RecordSink + Send), bytes_sink: Option>, ) { - let mut stream = WalStream::new(1, SEG_SIZE, Pos::ZERO).unwrap(); + let mut stream = WalStream::builder(1, SEG_SIZE, Pos::ZERO).unwrap(); if let Some(bs) = bytes_sink { stream.set_bytes_sink(bs); } let mut seg_sink = CollectingSegmentSink::default(); let start = Instant::now(); + let mut stream = stream.start(); for i in 0..iterations { let lsn = (i as u64) * SEG_SIZE; stream @@ -324,7 +325,6 @@ async fn pump_throughput_breakdown() { next_lsn: r.next_lsn, page_magic: r.page_magic, route: r.route, - catalog_boundary: r.catalog_boundary, boundary_info: r.boundary_info.clone(), aborted_tree: r.aborted_tree.clone(), defer_catalog_decode: r.defer_catalog_decode, diff --git a/tests/xact_buffer.rs b/tests/xact_buffer.rs index 56b4b8ca..ddcef6dc 100644 --- a/tests/xact_buffer.rs +++ b/tests/xact_buffer.rs @@ -39,7 +39,7 @@ use walshadow::shadow_catalog::{ShadowCatalog, ShadowCatalogConfig}; use walshadow::spill::ToastChunk; use walshadow::toast::{ChunkRefMap, MemChunkStore, ToastResolver}; use walshadow::xact_buffer::{ - WalkStep, XactBuffer, XactBufferConfig, XactBufferError, detoast_heap, + StashResolved, WalkStep, XactBuffer, XactBufferConfig, XactBufferError, detoast_heap, }; fn make_shadow(tmp: &tempfile::TempDir, port: u16) -> Shadow { @@ -229,14 +229,14 @@ async fn log_from_catalog(cat: &Arc>, dir: &std::path::Path async fn drain_all( b: &mut XactBuffer, resolver: &ToastResolver, - xid: u32, + stash: StashResolved, commit_ts: i64, commit_lsn: u64, subxids: &[u32], ) -> Result, XactBufferError> { let mut drain = b .drain_committed( - xid, + stash, commit_ts, commit_lsn, subxids, @@ -323,9 +323,16 @@ async fn commit_drains_in_arrival_order_and_clears_state() { )) .await .unwrap(); - let seen = drain_all(&mut b, &ToastResolver::disabled(), 7, 12345, 300, &[]) - .await - .unwrap(); + let seen = drain_all( + &mut b, + &ToastResolver::disabled(), + StashResolved::nothing_stashed(7), + 12345, + 300, + &[], + ) + .await + .unwrap(); assert_eq!(seen.len(), 2); assert_eq!(seen[0].decoded.source_lsn, 100); assert_eq!(seen[1].decoded.source_lsn, 200); @@ -389,7 +396,6 @@ async fn defer_catalog_decode_stashes_raw_and_commit_fences() { next_lsn: 160, page_magic: 0xD116, route: Route::ToDecoder, - catalog_boundary: false, boundary_info: None, aborted_tree: None, defer_catalog_decode: true, @@ -409,7 +415,7 @@ async fn defer_catalog_decode_stashes_raw_and_commit_fences() { "defer path, not marker path" ); let stats = Arc::new(EmitterStats::default()); - resolve_stash( + let stash = resolve_stash( &buffer, &log, &Default::default(), @@ -421,7 +427,7 @@ async fn defer_catalog_decode_stashes_raw_and_commit_fences() { .await .unwrap(); let mut b = buffer.lock().await; - let err = drain_all(&mut b, &ToastResolver::disabled(), 77, 0, 1000, &[]) + let err = drain_all(&mut b, &ToastResolver::disabled(), stash, 0, 1000, &[]) .await .expect_err("payload-less raw record must fail closed, not skip"); assert!( @@ -445,8 +451,11 @@ async fn commit_unknown_xid_no_ops() { // Even with no buffered records the commit's source LSN advances // `drain_lsn`, and the caller still registers a seq (had_states=false) // so the contiguous watermark passes read-only / filter-dropped xacts. - let mut drain = b.drain_committed(99, 0, 0x9000, &[], false).await.unwrap(); - assert!(!drain.had_states); + let mut drain = b + .drain_committed(StashResolved::nothing_stashed(99), 0, 0x9000, &[], false) + .await + .unwrap(); + assert!(!drain.had_states()); assert!( drain .next_batch(usize::MAX, usize::MAX, None) @@ -496,9 +505,16 @@ async fn commit_drains_spilled_then_in_memory_entries() { .await .unwrap(); } - let seen = drain_all(&mut b, &ToastResolver::disabled(), 5, 0, 250, &[]) - .await - .unwrap(); + let seen = drain_all( + &mut b, + &ToastResolver::disabled(), + StashResolved::nothing_stashed(5), + 0, + 250, + &[], + ) + .await + .unwrap(); assert_eq!(seen.len(), 5); for (i, c) in seen.iter().enumerate() { let lsn = c.decoded.source_lsn; @@ -544,9 +560,16 @@ async fn commit_merges_top_and_subxact_in_source_lsn_order() { b.on_heap(described(&log, heap(rfn, 7, 200, HeapOp::Insert, col(3)))) .await .unwrap(); - let seen = drain_all(&mut b, &ToastResolver::disabled(), 7, 12345, 300, &[8]) - .await - .unwrap(); + let seen = drain_all( + &mut b, + &ToastResolver::disabled(), + StashResolved::nothing_stashed(7), + 12345, + 300, + &[8], + ) + .await + .unwrap(); let lsns: Vec = seen.iter().map(|c| c.decoded.source_lsn).collect(); assert_eq!(lsns, [100, 150, 200]); // Per-top accounting: one bump, regardless of subxact count. @@ -604,9 +627,16 @@ async fn detoast_concatenates_uncompressed_chunks_into_text() { )) .await .unwrap(); - let seen = drain_all(&mut b, &ToastResolver::disabled(), 33, 12345, 300, &[]) - .await - .unwrap(); + let seen = drain_all( + &mut b, + &ToastResolver::disabled(), + StashResolved::nothing_stashed(33), + 12345, + 300, + &[], + ) + .await + .unwrap(); assert_eq!(seen.len(), 1); let body_col = &seen[0].decoded.new.as_ref().unwrap().columns[1]; match body_col { @@ -667,9 +697,16 @@ async fn detoast_missing_chunk_seq_errors_clearly() { Arc::new(MemChunkStore::new()), Arc::new(EmitterStats::default()), ); - let err = drain_all(&mut b, &resolver, 42, 0, 200, &[]) - .await - .expect_err("missing chunk surfaces"); + let err = drain_all( + &mut b, + &resolver, + StashResolved::nothing_stashed(42), + 0, + 200, + &[], + ) + .await + .expect_err("missing chunk surfaces"); match err { XactBufferError::MissingToastChunk { value_id, missing, ..