From 480bd765cb1c04a908699324c7ca8b7223e0f688 Mon Sep 17 00:00:00 2001 From: Almog Gavra Date: Wed, 29 Jul 2026 10:02:22 -0700 Subject: [PATCH 01/65] add distributed compaction blog post (#1929) --- .../distributed-compaction-scaling.html | 133 ++++++ .../public/charts/subcompaction-scaling.html | 169 +++++++ website/src/components/PostToc.astro | 2 +- website/src/content.config.ts | 3 + .../src/content/blog/compaction-roadmap.mdx | 437 ++++++++++++++++++ website/src/pages/blog/[...slug].astro | 4 +- 6 files changed, 745 insertions(+), 3 deletions(-) create mode 100644 website/public/charts/distributed-compaction-scaling.html create mode 100644 website/public/charts/subcompaction-scaling.html create mode 100644 website/src/content/blog/compaction-roadmap.mdx diff --git a/website/public/charts/distributed-compaction-scaling.html b/website/public/charts/distributed-compaction-scaling.html new file mode 100644 index 0000000000..b2e506de69 --- /dev/null +++ b/website/public/charts/distributed-compaction-scaling.html @@ -0,0 +1,133 @@ + + + + + +SlateDB — distributed compaction scaling (workers vs throughput, RFC-0025) + + + + +
+

Throughput scales with workers

+
+
+
+
+ + + + + + + diff --git a/website/public/charts/subcompaction-scaling.html b/website/public/charts/subcompaction-scaling.html new file mode 100644 index 0000000000..44af0cc5de --- /dev/null +++ b/website/public/charts/subcompaction-scaling.html @@ -0,0 +1,169 @@ + + + + + +SlateDB — subcompaction scaling (16 GB compaction, RFC-0028) + + + + +
+

Near-linear speedup

+
+
+ +

Wall-clock compaction time

+
+
+
+
+ + + + + + + diff --git a/website/src/components/PostToc.astro b/website/src/components/PostToc.astro index b089a484ef..8b0f661df7 100644 --- a/website/src/components/PostToc.astro +++ b/website/src/components/PostToc.astro @@ -6,7 +6,7 @@ interface Props { } const { headings } = Astro.props; -const items = headings.filter((h) => h.depth >= 2 && h.depth <= 4); +const items = headings.filter((h) => h.depth >= 2 && h.depth <= 3); --- {items.length > 0 && ( diff --git a/website/src/content.config.ts b/website/src/content.config.ts index c23728a710..af37251b23 100644 --- a/website/src/content.config.ts +++ b/website/src/content.config.ts @@ -17,6 +17,9 @@ export const collections = { authorGithub: z.string().optional(), // Optional per-post social image; falls back to the site default. ogImage: z.string().optional(), + // Optional canonical URL for cross-posted/syndicated content; points + // search engines at the original. Falls back to this page's own URL. + canonicalUrl: z.string().url().optional(), }), }), }; diff --git a/website/src/content/blog/compaction-roadmap.mdx b/website/src/content/blog/compaction-roadmap.mdx new file mode 100644 index 0000000000..882c8ff431 --- /dev/null +++ b/website/src/content/blog/compaction-roadmap.mdx @@ -0,0 +1,437 @@ +--- +title: The Past, Present, and Future of Compaction in SlateDB +pubDate: 2026-07-13 +author: Ryan Dielhenn, Almog Gavra +canonicalUrl: https://ryandielhenn.github.io/blog/distributed-compaction/ +--- + +import ChartEmbed from '../../components/ChartEmbed.astro'; + +[SlateDB](https://slatedb.io) is an embedded key-value store built on object +storage. + +SlateDB uses a Log-Structured Merge Tree, or LSM for short, to batch writes to +object storage. Incoming writes land in an in-memory buffer called the +memtable. Once the memtable fills up, it is flushed to an immutable file stored +in object storage called a Sorted String Table (SST). + +LSM trees are structured in multiple levels, and the first time an SST lands +in object store it is put in the first level known as L0. From there, SlateDB +uses a process called compaction, where SSTs are grouped and merged together as +each tier fills up. + +LSM trees let you tune the tradeoffs between read, write, and space +amplification. I recommend reading [this +blog](https://www.bitsxpages.com/p/understanding-lsm-trees-via-read) by Almog +Gavra if you want to know more about these tradeoffs. + +The following is what you would see if you listed the contents of an object +storage bucket path used by SlateDB: + +```text +manifest/ + 00000000001.manifest # This is a snapshot of the database state: + 00000000002.manifest # SST lists, watermarks, epochs, external dbs, checkpoints etc. + 00000000003.manifest + ... +compactions/ + 00000000001.compactions # Jobs scheduled for compaction by the + 00000000002.compactions # compaction coordinator. + 00000000003.compactions + ... +compacted/ + .sst # This is compacted data + .sst # L0 ssts also happen to live here which have not been compacted yet + .sst + ... +wal/ # This is the write ahead log. Writes land here first so that they can be replayed in a failure scenario. + 00000000001.sst + 00000000002.sst + 00000000003.sst +gc/ + manifest.boundary # Garbage collector deletes .manifest versions at or below this Boundary + compactions.boundary # Garbage collector deletes .compactions files at or below this Boundary +``` + +All of this is hidden from the user under simple put/get/scan APIs. + +## Compaction + +Compaction is a critical background process of the LSM tree that takes Sorted +String Tables (SST for short) and merges them to produce an output SST with +non-repeating keys. + +This process does a few things. When multiple SSTs share keys, merging them +removes duplicate entries and cleans up tombstones left behind by deletes, +reducing space amplification. It also reduces the number of SSTs that need to +be read to find a key by improving the locality of sorted data into longer +runs: + +```ascii-art +┌SST A · newest────────────┐ ┌SST B─────────────────────┐ ┌SST C · oldest────────────┐ +│orders:1042 @14 shipped │ │orders:1042 @8 packed │ │orders:1042 @3 placed │ +│orders:1058 @13 paid │ │orders:1058 @7 pending │ │orders:1023 @2 paid │ +│orders:1023 @12 ⌫ deleted│ │orders:1023 @6 shipped │ │users:42 @1 a@old.co │ +│users:42 @11 a@new.co │ │users:91 @5 plan:pro │ │users:91 @0 plan:free│ +└──────────────────────────┘ └──────────────────────────┘ └──────────────────────────┘ + + │ │ │ + └──────────────────────────────┼──────────────────────────────┘ + ▼ + ┌k-way merge · min-heap────────────────────┐ + │pop smallest (key, -LSN) across streams │ + └──────────────────────────────────────────┘ + ▼ + ┌per-key winner · highest LSN wins───────────────────┐ + │orders:1042 keep @14 shipped · drop @8, @3 │ + │orders:1023 ⌫ tombstone @12 · drop @6, @2 │ + │users:42 keep @11 a@new.co · drop @1 │ + │orders:1058 keep @13 paid · drop @7 │ + │users:91 keep @5 plan:pro · drop @0 │ + └────────────────────────────────────────────────────┘ + ▼ + ┌output SST · one current row per key────────────────┐ + │orders:1042 → shipped users:42 → a@new.co │ + │orders:1058 → paid users:91 → plan:pro │ + │orders:1023 → ⌫ dropped (tombstone removed) │ + └────────────────────────────────────────────────────┘ +``` + +

The LSM compaction merge step: a k-way merge keeps each +key's highest-LSN version and drops the rest. Rows are written key @LSN → +value; ⌫ marks a tombstone (a deleted key).

+ +## Distributed Compaction + +A single compactor is a bottleneck: if it cannot keep pace with write +throughput, the whole system degrades in two stages: + +1. as uncompacted SSTs pile up in L0, more files need to be scanned to + find a key, increasing read latency. +2. once the L0 file count reaches `l0_max_ssts`, the flusher stops writing + immutable memtables to L0. Those memtables accumulate in memory until + `max_unflushed_bytes` is exceeded, at which point SlateDB applies + backpressure that stalls writes from being durably written to object + storage. + +A lagging compactor therefore degrades read latency first, then write throughput. + +In 0.14.1 we introduced distributed compaction, which mitigates this +issue by allowing SlateDB to leverage multiple concurrent workers +to run compactions. + +```ascii-art + Single compactor │ Distributed workers + ┌write path · flushes fast─────────┐ │ ┌write path · flushes fast─────────┐ + └──────────────────────────────────┘ │ └──────────────────────────────────┘ + ▼ │ ▼ + ┌L0 SST count rising───────────────┐ │ ┌compaction keeps up───────────────┐ + │█ █ █ █ █ █ █ █ █ █ █ █ │ │ │█ █ █ ░ ░ ░ ░ ░ ░ ░ ░ ░ │ + │read latency degrades │ │ │drained as fast as it fills │ + └──────────────────────────────────┘ │ └──────────────────────────────────┘ + ▼ │ ▼ + ┌flusher stops at l0_max_ssts──────┐ │ ┌.compactions · job queue──────────┐ + │immutable memtables pile in RAM │ │ └──────────────────────────────────┘ + └──────────────────────────────────┘ │ ┌────────────┬────────────┐ + ▼ │ ▼ ▼ ▼ + ┌max_unflushed_bytes exceeded──────┐ │ ┌worker 1──┐ ┌worker 2──┐ ┌worker N──┐ + │back-pressure stalls writes │ │ │job a, b │ │job c, d │ │job e, f │ + └──────────────────────────────────┘ │ └──────────┘ └──────────┘ └──────────┘ + ▼ │ │ │ │ + ┌1 compactor draining──────────────┐ │ └────────────┼────────────┘ + │capped by one node's CPU + net │ │ ▼ merged SSTs + └──────────────────────────────────┘ │ ┌sorted runs───────────────────────┐ + │ │█ █ █ █ █ █ █ █ │ +A lagging compactor degrades the │ └──────────────────────────────────┘ +whole system: reads slow first, │ +then writes stall. │ Add workers to raise the ceiling; + │ the single-writer manifest + │ invariant is preserved. +``` + +

How distributing compaction across stateless workers relieves the single-compactor bottleneck.

+ +Distributed compaction works by designating a single coordinator for compactions +that handles scheduling. This coordinator writes the results of scheduling into +a shared [transactional +object](https://github.com/slatedb/slatedb/blob/main/slatedb-txn-obj/README.md) +on object storage called the `.compactions` file, and workers poll this file to +claim compactions, updating that file when they have made progress on executing +a transaction. + +To understand its effect on scaling we ran a benchmark that clears out a backlog +of independent per-segment ~2GB compactions while varying the number of +workers, each pinned to a single vCPU. Since the jobs are independent and each +runs on its own worker, throughput scales near-linearly all the way to six +workers: 5.2× the single-worker rate while holding above 85% parallel +efficiency. + + + +In addition to allowing for improved compaction throughput, distributed compaction +enables some nice quality of life operational improvements: + +1. You can migrate from running compaction embedded on the writer to running it + standalone (with one or more workers) without fencing the active writer. +2. You can start/stop compactor workers without worrying about fencing at all. +3. You can run compactors on a compute framework like kubernetes without + installing a separate controller. + +The full design of distributed compactions is in +[RFC-0025](/rfcs/0025-distributed-compaction) and is well worth a read. + +## Future benefits and work + +The performance and operational improvements of distributed compaction are +just the start. The door is wide open for future enhancements that take +advantage of these stateless compaction workers. Below are just a few examples +of extensions made possible by the stateless workers added in RFC-0025. + +### Concurrent L0 Compactions + +Ideally we want to parallelize compaction work, but it helps to separate two +axes of parallelism that are easy to conflate. + +The first axis is running *independent* compactions at the same time. There are +two different types of independent compactions: + +1. Disjoint sorted run compactions. These are compactions that target already + compacted data from different levels of the tree (e.g. `L0→SR1` and + `[SR2,SR3]→SR4`). +2. Compactions from different segments. + [RFC-0024](/rfcs/0024-segment-oriented-compaction) + introduced segmented compaction which allows SlateDB to maintain multiple + LSM trees that share a single WAL/memtable. Compactions that target + different segments are independent, even targeting L0. + +Distributed compaction is what lets you execute these across more than one +machine: a single embedded worker is capped by one machine's CPU and I/O, and +`max_concurrent_compactions` only stretches that one machine so far. Spreading +independent jobs across a pool of workers is the bottleneck relief this post is +about. + +The second axis is parallelizing L0 compactions within the same segment, and +here it's worth not overselling: distributed compaction does **not** unlock it +on its own. + +`last_compacted_l0_sst_view_id` is a single cursor over a segment's L0 list, +and "already compacted" means "at or below the cursor." A single monotonic +boundary can't represent two disjoint, in-flight L0 consumptions at once, so L0 +compaction within a tree is serialized whether that tree is served by one +embedded worker or a fleet of remote ones. RFC-24 calls this out directly: + +> Parallel L0 compaction within a single segment is a separate concern tied to +> the watermark's single-cursor design and is not addressed here. + +```ascii-art +segment L0 list · oldest ──────────────────▶ newest +┌────┐ ┌────┐ ┌────┐ ┌────┐ ┌────┐ ┌────┐ +│ L1 │ │ L2 │ │ L3 │ │ L4 │ │ L5 │ │ L6 │ +└────┘ └────┘ └────┘ └────┘ └────┘ └────┘ + ▲ + cursor = last_compacted_l0_sst_view_id + (at/below = consumed · above = live) + +┌Job A · consume {L3, L4}────┐ ┌Job B · consume {L5, L6}────┐ +│still running … │ │finishes first │ +└────────────────────────────┘ └────────────────────────────┘ + +One monotonic cursor can't hold two in-flight consumptions: B commits, +the cursor jumps past L6, and L3/L4 are GC'd while A is still running. +``` + +

A single monotonic cursor can't represent two disjoint, in-flight L0 consumptions in the same segment.

+ +That note is just as true after RFC-25. Moving work onto stateless workers +changes where a compaction runs, not whether one L0 compaction can be split +in two. + +Closing that gap takes one of two things (or both), and neither is distributed +compaction: + +#### Improvement 1: Subcompactions + +Subcompactions ([RFC-0028](/rfcs/0028-subcompactions)). +Rather than splitting the L0 list across multiple compactions (which the +watermark forbids), a subcompaction keeps it as one logical compaction and +splits the *key range* into sub-ranges that run in parallel. The parent +commits a single manifest update that advances the watermark exactly once over +the whole consumed set, so the single-cursor invariant is never violated which +sidesteps the problem instead of fighting it. + +Subcompactions parallelize across cores on one worker, which composes +cleanly with distributed compaction parallelizing across workers; allowing +separate workers to claim subcompactions of the same parent compaction is +explicitly labeled as future work in RFC-0028 and out of scope. + +```ascii-art +┌one logical compaction · live set {L3 … L6}─┐ +│┌────┐ ┌────┐ ┌────┐ ┌────┐ │ +││ L3 │ │ L4 │ │ L5 │ │ L6 │ │ +│└────┘ └────┘ └────┘ └────┘ │ +└────────────────────────────────────────────┘ + split by key range + ▼ ▼ +┌sub-range A · core 1┐ ┌sub-range B · core 2┐ +│keys (−∞, k) │ │keys [k, +∞) │ +│output SST(s) │ │output SST(s) │ +└────────────────────┘ └────────────────────┘ + │ │ + └───────────┬───────────┘ + ▼ +┌single manifest commit──────────────────────┐ +│advance cursor once: before L3 → after L6 │ +└────────────────────────────────────────────┘ +``` + +

Parallelism from splitting one compaction by key range; the cursor still advances exactly once per parent compaction.

+ +In practice, splitting a single 16 GB compaction into key-range subcompactions +scales almost linearly. The results below are from a benchmark running a 12-way +split ~11× faster than the baseline no-subcompactions run, collapsing a 2m37s +compaction to ~14s while holding above 90% parallel efficiency. + + + +

Benchmark evidence for subcompactions (RFC-0028): measured speedup tracks the ideal-linear reference, holding above 90% efficiency through a 12-way split.

+ +#### Improvement 2: Reworking the L0 Watermark + +Reworking the watermark to track a *set* of consumed L0 SSTs instead of a +single cursor. This is the more invasive change, but it's what would let two L0 +parent compactions in the same segment advance independently. + +```ascii-art +segment L0 list · oldest ──────────────────▶ newest +track a SET of consumed SSTs ( ● consumed ○ live ) +┌────┐ ┌────┐ ┌────┐ ┌────┐ ┌────┐ ┌────┐ +│L1 ●│ │L2 ●│ │L3 ○│ │L4 ○│ │L5 ○│ │L6 ○│ +└────┘ └────┘ └────┘ └────┘ └────┘ └────┘ + ▲ ▲ ▲ ▲ + └───┬───┘ └───┬───┘ + Job A Job B + {L3,L4} {L5,L6} + independent independent + +A set-valued watermark lets two L0 compactions in the same segment +advance independently so neither GC's the other's live inputs. +``` + +

Track a set of compacted SSTs instead of one boundary, so two L0 compactions in the same segment advance independently.

+ +To summarize, distributed compaction removes the restriction that compaction +runs on a single node, subcompactions remove the restriction that a single +compaction runs on a single CPU, and the watermark design (still to be +implemented as of July 2026) will allow us to run concurrent L0 compactions +within a single segment. + +### Priority Based Compaction Routing + +Once compactions are jobs claimed by stateless workers rather than steps run +inline by one process, the coordinator is free to decide *which* job goes +*where*. + +Compactions could be routed to specific workers, or pools of them, +based on priority: an L0 compaction on a segment approaching `l0_max_ssts` is +far more urgent than a routine sorted-run merge, because the former is what +stands between the database and write-stalling backpressure. High-priority jobs +could be steered to a dedicated set of low-latency workers while bulk +sorted-run merges run on cheaper, best-effort capacity. + +This will turn the worker pool into a scheduling surface where compaction +resources follow the work that is most likely to degrade read and write +latency. + +```ascii-art +┌.compactions · job queue with priority hints────────┐ +│L0 drain · hot large-tier merge · heavy │ +└────────────────────────────────────────────────────┘ + │ │ +┌router · reads priority, dispatches to matching pool┐ +└────────────────────────────────────────────────────┘ + ▼ ▼ +┌hot pool────────────────┐ ┌heavy pool──────────────┐ +│L0 drains · latency │ │big merges · thruput │ +│████ ████ ████ ████ │ │██████████ ██████████ │ +└────────────────────────┘ └────────────────────────┘ +``` + +

Priority hints let the coordinator route hot L0 drains and heavy sorted-run merges to matching worker pools.

+ +### Shared Worker Pools + +Because the workers are stateless, nothing about a compaction job ties it to a +single database instance. + +A shared worker pool serving many instances would significantly reduce the +I/O-bound threads each instance has to reserve for itself, and let instances +trade compaction resources as needed e.g. an idle database contributes its +share of the pool to a neighbor that is busy ingesting. Without it, capacity is +sized per database for that database's worst case. This is true of an embedded +compactor and equally of a remote fleet dedicated to one instance, unless you +build per-DB autoscaling. + +Pooling that capacity absorbs those bursts and raises overall utilization, +while priority-based routing decides how the shared pool is divided when +several instances contend for it at once. + +```ascii-art + Per-DB workers │ Shared workers +┌DB α────────┐ ┌DB β────────┐ ┌DB γ────────┐ │ ┌DB α────────┐ ┌DB β────────┐ ┌DB γ────────┐ +│ idle │ │ hot │ │ idle │ │ │ idle │ │ hot │ │ idle │ +└────────────┘ └────────────┘ └────────────┘ │ └────────────┘ └────────────┘ └────────────┘ + ▼ ▼ ▼ │ ▼ ▼ ▼ +┌pool α──────┐ ┌pool β──────┐ ┌pool γ──────┐ │ ┌.compactions┐ ┌.compactions┐ ┌.compactions┐ +│ ░ ░ ░ │ │ █ █ █ │ │ ░ ░ ░ │ │ │ α (empty) │ │ β (6 jobs) │ │ γ (1 job) │ +└────────────┘ └────────────┘ └────────────┘ │ └────────────┘ └────────────┘ └────────────┘ + │ └──────────────┼──────────────┘ +each DB reserves its own threads; │ ▼ +α and γ sit idle while β is saturated. │ ┌shared worker pool────────────────────────┐ + │ │polls every DB's .compactions │ +resources are locked per database — │ │ │ +β backs up but can't borrow α or γ. │ │ ┌───┬───┬───┬───┬───┬───┐ │ + │ │ │ β │ β │ β │ β │ γ │ ░ │ │ +you provision (and pay for) peak × N. │ │ └───┴───┴───┴───┴───┴───┘ │ + │ │ │ + │ │box = worker · β/γ = its DB · ░ = idle │ + │ └──────────────────────────────────────────┘ + │ + │ capacity follows demand: + │ β gets 4 · γ gets 1 · α gets 0 · 1 spare. + │ + │ pay for aggregate peak, not N × peak — + │ the spare flows wherever it is needed. +``` + +

A shared worker pool serving multiple SlateDB instances: capacity follows demand instead of being pinned per database.

+ +## Conclusion + +Distributed compaction (RFC-0025) removes the single-*process* ceiling on +compaction and lays down a stateless-worker foundation: jobs are claimed and +executed by workers that hold no durable state of their own. + +On its own that relieves the single-compactor bottleneck that degrades read +latency and then write throughput while improving the operational quality of +life of compaction in SlateDB. + +Just as importantly, the stateless-worker model is what makes the future work +above tractable. Reworking the watermark design, adding priority-based routing +and shared worker pool support are natural extensions now that a compaction is +just a job that any worker can claim. + +:::note[attribution] +This post was originally published on Ryan Dielhenn's blog at +[ryandielhenn.github.io](https://ryandielhenn.github.io/blog/distributed-compaction/), +and reproduced here with permission. This version has modified some text and added +additional benchmarking results. +::: + diff --git a/website/src/pages/blog/[...slug].astro b/website/src/pages/blog/[...slug].astro index e846aa874e..3d6d51ff13 100644 --- a/website/src/pages/blog/[...slug].astro +++ b/website/src/pages/blog/[...slug].astro @@ -10,7 +10,7 @@ export async function getStaticPaths() { const { post } = Astro.props; const { Content, headings } = await render(post); -const { title, author, authorGithub, pubDate, ogImage } = post.data; +const { title, author, authorGithub, pubDate, ogImage, canonicalUrl } = post.data; // Default the social card to the per-post image generated at build time // (src/pages/blog/og/[...slug].png.ts); frontmatter `ogImage` can override it. @@ -25,7 +25,7 @@ const dateLabel = pubDate.toLocaleDateString('en-US', { const isoDate = pubDate.toISOString(); --- - +
From 3428d06357684f0bbe82682de92ecbd0228f8035 Mon Sep 17 00:00:00 2001 From: Rohan Date: Wed, 29 Jul 2026 19:13:59 -0400 Subject: [PATCH 02/65] [rfc-30 2/N]: add traits for WAL iteration and update replay paths (#1976) --- rfcs/0030-pluggable-wal.md | 14 +- slatedb/src/db.rs | 38 +- slatedb/src/db/builder.rs | 4 +- slatedb/src/db_reader.rs | 66 +-- slatedb/src/error.rs | 6 +- slatedb/src/fence.rs | 32 +- slatedb/src/wal/mod.rs | 75 ++- slatedb/src/wal/writer_init.rs | 22 +- slatedb/src/wal_buffer.rs | 2 +- slatedb/src/wal_replay.rs | 807 +++++++++++++++++++++++++++------ 10 files changed, 789 insertions(+), 277 deletions(-) diff --git a/rfcs/0030-pluggable-wal.md b/rfcs/0030-pluggable-wal.md index b18081daaa..94e71c07b3 100644 --- a/rfcs/0030-pluggable-wal.md +++ b/rfcs/0030-pluggable-wal.md @@ -347,14 +347,8 @@ pub struct WalRows { /// The rows read from the WAL File. All the rows with a given sequence number must be present /// in th same [`WalRows`]. pub rows: Vec, - /// The id of the last WAL File containing rows from `rows`. There may still be rows with higher - /// sequence numbers in the WAL File with this id. - pub last_wal_file_id: u64, - /// True when this batch is the last one in its WAL file. This is an - /// optimization, so its harmless to always set to false. Callers can already infer that a - /// file is fully applied when they see a batch from a later file, but this flag lets them - /// advance their WAL watermark over the current file without waiting for the next one. - pub last_in_file: bool, + /// The id of the last WAL File for which all rows have been consumed by the iterator. + pub last_consumed_wal_file_id: u64, } /// An iterator over rows in some range of the WAL @@ -714,8 +708,8 @@ implementation is correct. Some important test cases we'll cover (non-exhaustive - `WalWriter` emits events when rows are durably stored. - `WalIterator` always iterates over writes in sequence order - `WalIterator` always returns full write batches in `WalRows` -- `WalIterator` tracks the WAL file id in `WalRows` correctly (TODO: this probably needs some - test interfaces in the reader for listing/reading wal files) +- `WalIterator` tracks the last consumed WAL file id in `WalRows` correctly (TODO: this probably + needs some test interfaces in the reader for listing/reading wal files) **Performance** diff --git a/slatedb/src/db.rs b/slatedb/src/db.rs index 31699c192f..df23172470 100644 --- a/slatedb/src/db.rs +++ b/slatedb/src/db.rs @@ -23,8 +23,6 @@ pub use crate::db_status::{DbStatus, SegmentPrefix}; use crate::db_cache::CacheTarget; -use crate::db_cache_manager; -use std::ops::Range; use std::sync::Arc; use bytes::Bytes; @@ -37,7 +35,7 @@ use crate::db_transaction::DbTransaction; use crate::dispatcher::MessageHandlerExecutor; use crate::garbage_collector::GC_TASK_NAME; use crate::transaction_manager::IsolationLevel; -use crate::CloseReason; +use crate::{db_cache_manager, CloseReason}; use log::{debug, info, trace, warn}; use parking_lot::RwLock; use std::time::Duration; @@ -57,7 +55,6 @@ use crate::db_snapshot::DbSnapshot; use crate::db_state::{collect_touched_segments, DbState, SsTableId}; use crate::db_stats::DbStats; use crate::error::SlateDBError; -use crate::iter::IterationOrder; use crate::manifest::{Manifest, VersionedManifest}; use crate::mem_table::KVTableMetadata; use crate::memtable_flusher::{FlushResult, FlushTarget, MemtableFlusher}; @@ -67,7 +64,6 @@ use crate::paths::PathResolver; use crate::prefix_extractor::PrefixExtractor; use crate::reader::{Reader, ScanContext}; use crate::snapshot_manager::SnapshotManager; -use crate::sst_iter::SstIteratorOptions; use crate::tablestore::TableStore; use crate::transaction_manager::TransactionManager; use crate::types::KeyValue; @@ -80,7 +76,7 @@ use slatedb_common::DbRand; use slatedb_txn_obj::DirtyObject; use crate::db_status::{ClosedResultWriter, DbStatusManager}; -use crate::wal::{WalEvent, WalObserver, WalStatus}; +use crate::wal::{WalEvent, WalIterator, WalObserver, WalStatus}; pub use builder::DbBuilder; pub use builder::DbReaderBuilder; @@ -477,7 +473,7 @@ impl DbInner { } } - async fn replay_wal(&self, wal_id_range: Range) -> Result<(), SlateDBError> { + async fn replay_wal(&self, wal_iterator: Box) -> Result<(), SlateDBError> { let mut current_memtable_wal_id = self .state .read() @@ -494,32 +490,18 @@ impl DbInner { |_| -> Result<(), SlateDBError> { Ok(()) } ); - let sst_iter_options = SstIteratorOptions { - max_fetch_tasks: 1, - blocks_to_fetch: 256, - cache_blocks: false, - cache_metadata: false, - eager_spawn: true, - order: IterationOrder::Ascending, - prefix: None, - filter_context: None, - }; - let replay_options = WalReplayOptions { - sst_batch_size: 4, max_memtable_bytes: self.settings.l0_sst_size_bytes, - sst_iter_options, - min_seq: None, + ..Default::default() }; let db_state = self.state.read().state().core().clone(); - let mut replay_iter = WalReplayIterator::range( - wal_id_range, + let mut replay_iter = WalReplayIterator::for_wal_iterator( + wal_iterator, &db_state, replay_options, Arc::clone(&self.table_store), - ) - .await?; + )?; loop { let replayed_table = match replay_iter.next().await { @@ -529,12 +511,12 @@ impl DbInner { // indicate that a newer writer has advanced `replay_after_wal_id` and // the GC has removed this WAL entry. Check the latest manifest's // writer_epoch to see if this client is fenced. - Err(err) if err.has_object_store_not_found() => { + Err(SlateDBError::WalTruncated(wal_id)) => { self.memtable_flusher.refresh_manifest().await?; if self.state.read().state().manifest.value.writer_epoch > writer_epoch { return Err(SlateDBError::Fenced); } - return Err(err); + return Err(SlateDBError::WalTruncated(wal_id)); } Err(err) => return Err(err), }; @@ -2167,7 +2149,7 @@ mod tests { REQUEST_COUNT as OBJECT_STORE_REQUEST_COUNT, REQUEST_DURATION_SECONDS as OBJECT_STORE_REQUEST_DURATION_SECONDS, }; - use crate::iter::RowEntryIterator; + use crate::iter::{IterationOrder, RowEntryIterator}; use crate::manifest::store::{ManifestStore, StoredManifest}; use crate::manifest::{ManifestCore, VersionedManifest}; use crate::merge_operator::{ diff --git a/slatedb/src/db/builder.rs b/slatedb/src/db/builder.rs index 741e9a5024..e427e56aad 100644 --- a/slatedb/src/db/builder.rs +++ b/slatedb/src/db/builder.rs @@ -617,7 +617,7 @@ impl> DbBuilder

{ ); let WriterFenceResult { manifest, - replay_range, + replay_iterator, mut wal_writer, } = fencer.fence(stored_manifest).await?; let (wal_writer, wal_observer) = if DbInner::wal_enabled_in_options(&self.settings) { @@ -818,7 +818,7 @@ impl> DbBuilder

{ task_executor.monitor_on(&tokio_handle)?; // Replay WAL - inner.replay_wal(replay_range).await?; + inner.replay_wal(replay_iterator).await?; // Preload cache if enabled if let Some(cached_obj_store) = cached_object_store { diff --git a/slatedb/src/db_reader.rs b/slatedb/src/db_reader.rs index 5f0aed5ced..734dee4387 100644 --- a/slatedb/src/db_reader.rs +++ b/slatedb/src/db_reader.rs @@ -23,7 +23,7 @@ use crate::sst_iter::SstIteratorOptions; use crate::tablestore::TableStore; use crate::types::KeyValue; use crate::utils::IdGenerator; -use crate::wal_replay::{WalReplayIterator, WalReplayOptions}; +use crate::wal_replay::{WalIteratorOptions, WalReplayIterator, WalReplayOptions}; use crate::{Checkpoint, DbIterator}; use crate::{DbCacheManagerOps, DbMetadataOps, DbReadOps}; use async_trait::async_trait; @@ -611,10 +611,12 @@ impl DbReaderInner { core.next_wal_sst_id }; - let replay_options = WalReplayOptions { + let iterator_options = WalIteratorOptions { sst_batch_size: 4, - max_memtable_bytes: reader_options.max_memtable_bytes as usize, sst_iter_options, + }; + let replay_options = WalReplayOptions { + max_memtable_bytes: reader_options.max_memtable_bytes as usize, // Skip entries that we already have in `imm_memtable` (that might be above last_l0_seq). min_seq: Some(last_committed_seq), }; @@ -622,18 +624,21 @@ impl DbReaderInner { let mut replay_iter = WalReplayIterator::range( (replay_after_wal_id + 1)..wal_id_end, core, + iterator_options, replay_options, Arc::clone(&table_store), - ) - .await?; + )?; while let Some(replayed_table) = match replay_iter.next().await { Ok(Some(replayed_table)) => Some(replayed_table), Ok(None) => None, - Err(err) if has_not_found_object_store_error(&err) => None, + Err(SlateDBError::WalTruncated(_)) => None, Err(err) => return Err(err), } { - assert!(replayed_table.last_wal_id > replay_after_wal_id); + // `last_wal_id` is a conservative watermark: a table that ends mid-file + // is tagged with the last fully replayed WAL ID, which may equal the + // watermark of the previous table. + assert!(replayed_table.last_wal_id >= replay_after_wal_id); replay_after_wal_id = replayed_table.last_wal_id; if !replayed_table.table.is_empty() && replayed_table.last_seq > last_committed_seq { let first_seq = replayed_table @@ -1366,24 +1371,6 @@ impl DbCacheManagerOps for DbReader { } } -/// Checks if the error or any of its sources is an `object_store::Error::NotFound` error. -fn has_not_found_object_store_error(err: &(dyn std::error::Error + 'static)) -> bool { - let mut current = Some(err); - while let Some(current_err) = current { - if current_err - .downcast_ref::() - .is_some_and(|err| matches!(err, object_store::Error::NotFound { .. })) - || current_err - .downcast_ref::>() - .is_some_and(|err| matches!(err.as_ref(), object_store::Error::NotFound { .. })) - { - return true; - } - current = current_err.source(); - } - false -} - #[cfg(test)] mod tests { use super::{DbReaderMessage, ManifestPoller, ReaderState}; @@ -2289,26 +2276,6 @@ mod tests { .await; } - #[test] - fn has_not_found_object_store_error_should_walk_nested_error_sources() { - let err = crate::Error::from(SlateDBError::from(object_store::Error::NotFound { - path: "missing-wal".to_string(), - source: Box::new(std::io::Error::other("missing")), - })); - - assert!(super::has_not_found_object_store_error(&err)); - } - - #[test] - fn has_not_found_object_store_error_should_ignore_non_not_found_errors() { - let err = SlateDBError::from(object_store::Error::NotImplemented { - operation: "test".to_string(), - implementer: "test".to_string(), - }); - - assert!(!super::has_not_found_object_store_error(&err)); - } - #[tokio::test] async fn replay_wal_into_should_treat_missing_wal_sst_as_end_of_iteration() { let object_store: Arc = Arc::new(InMemory::new()); @@ -2339,9 +2306,12 @@ mod tests { .await .unwrap(); - assert_eq!(last_wal_id, 0); - assert_eq!(last_committed_seq, 0); - assert!(into_tables.is_empty()); + // WAL 2 is missing and ends the iteration, but the rows already replayed + // from WAL 1 must still be returned. + assert_eq!(last_wal_id, 1); + assert_eq!(last_committed_seq, 1); + assert_eq!(into_tables.len(), 1); + assert_eq!(into_tables.front().unwrap().recent_flushed_wal_id(), 1); } #[tokio::test] diff --git a/slatedb/src/error.rs b/slatedb/src/error.rs index ea5802b357..1ade60c558 100644 --- a/slatedb/src/error.rs +++ b/slatedb/src/error.rs @@ -77,8 +77,8 @@ pub(crate) enum SlateDBError { #[error("wal store reconfiguration unsupported")] WalStoreReconfigurationError, - #[error("wal truncated")] - WalTruncated, + #[error("wal truncated at wal file `{0}`")] + WalTruncated(u64), #[error("wal unavailable")] WalUnavailable(Arc), @@ -727,6 +727,7 @@ impl From for Error { SlateDBError::CloneExternalDbMissing => Error::data(msg), SlateDBError::CloneIncorrectExternalDbCheckpoint { .. } => Error::data(msg), SlateDBError::CloneIncorrectFinalCheckpoint { .. } => Error::data(msg), + SlateDBError::WalTruncated(_) => Error::data(msg), SlateDBError::WalDataError(src) => Error::data(msg).with_source(Box::new(src)), // Internal errors @@ -742,7 +743,6 @@ impl From for Error { SlateDBError::TransactionalObjectError(err) => { Error::internal(msg).with_source(Box::new(err)) } - SlateDBError::WalTruncated => Error::internal(msg), SlateDBError::WalInternalError(src) => Error::internal(msg).with_source(Box::new(src)), } } diff --git a/slatedb/src/fence.rs b/slatedb/src/fence.rs index ead82336e3..80e4754664 100644 --- a/slatedb/src/fence.rs +++ b/slatedb/src/fence.rs @@ -4,13 +4,11 @@ use crate::manifest::store::{FenceableManifest, StoredManifest}; use crate::tablestore::TableStore; use crate::utils::WatchableOnceCellReader; use crate::wal::writer_init::{WalWriterInit, WalWriterInitOptions}; -use crate::wal::{WalWriter, WriterInit}; +use crate::wal::{WalIterator, WalWriter, WriterInit}; use crate::Settings; use fail_parallel::{fail_point_send, FailPointTx}; -use log::error; use slatedb_common::metrics::MetricsRecorderHelper; use slatedb_common::SystemClock; -use std::ops::Range; use std::sync::Arc; use std::time::Duration; @@ -28,7 +26,7 @@ pub(crate) struct WriterFencer { pub(crate) struct WriterFenceResult { pub(crate) manifest: FenceableManifest, - pub(crate) replay_range: Range, + pub(crate) replay_iterator: Box, pub(crate) wal_writer: Box, } @@ -80,8 +78,8 @@ impl WriterFencer { /// Fences all writers with an older epoch than the provided `stored_manifest` by (1) writing /// a new `FenceableManifest` with a bumped epoch, and (2) writing an empty WAL file that acts /// as a barrier. Any parallel old writers will fail with `SlateDBError::Fenced` when trying - /// to "re-write" this file. Returns a `WriterFence` with the `FenceableManifest` and range - /// that must be replayed to recover up to the current epoch + /// to "re-write" this file. Returns a `WriterFence` with the `FenceableManifest` and iterator + /// that must be replayed to recover up to the current epoch. pub(crate) async fn fence( self, stored_manifest: StoredManifest, @@ -113,17 +111,10 @@ impl WriterFencer { manifest.refresh().await?; fail_point_send!(self.fp_tx, "FinalRefreshManifest"); - let replay_range = match result.replay_range.try_into() { - Ok(replay_range) => replay_range, - Err(_) => { - error!("replay range must use inclusive lower bound and exclusive upper bound"); - return Err(SlateDBError::InvalidDBState); - } - }; Ok(WriterFenceResult { manifest, wal_writer: result.wal_writer, - replay_range, + replay_iterator: result.replay_iterator, }) } } @@ -586,10 +577,19 @@ mod tests { // unpause WriterFencer fail_parallel::cfg(h.fp_registry.clone(), case.event, "off").unwrap(); // verify it returns successfully - let result = jh.await.unwrap().unwrap(); + let mut result = jh.await.unwrap().unwrap(); // The fencer's stale empty_wal_id was retried above the fenced writer's possibly // advanced replay_after_wal_id. - assert_eq!(result.replay_range.start, replay_after_wal_id + 1); + let first_replayed_wal = result + .replay_iterator + .next() + .await + .unwrap() + .expect("expected the replay iterator to contain the fencing WAL"); + assert_eq!( + first_replayed_wal.last_consumed_wal_file_id, + replay_after_wal_id + 1 + ); // verify that fenced db is fenced (new write fails) use crate::error::{CloseReason, ErrorKind}; diff --git a/slatedb/src/wal/mod.rs b/slatedb/src/wal/mod.rs index e709227412..eff19febab 100644 --- a/slatedb/src/wal/mod.rs +++ b/slatedb/src/wal/mod.rs @@ -15,7 +15,7 @@ pub(crate) mod wal_sst_builder; pub(crate) mod writer_init; /// A range of WAL File IDs -pub struct WalFileRange(Bound, Bound); +pub struct WalFileRange(pub Bound, pub Bound); impl From> for WalFileRange { fn from(range: Range) -> Self { @@ -41,7 +41,7 @@ pub enum WalError { /// The WAL writer was fenced Fenced, /// A WalIterator observed that the tail of the WAL was truncated while iterating. - WalTruncated, + WalTruncated(u64), /// Operation against wal after it was closed Closed, /// WAL is unavailable, e.g. due to an I/O error or error in the backing storage system @@ -56,7 +56,7 @@ impl Display for WalError { fn fmt(&self, f: &mut Formatter<'_>) -> std::fmt::Result { match self { WalError::Fenced => write!(f, "WAL writer was fenced"), - WalError::WalTruncated => write!(f, "WAL was truncated"), + WalError::WalTruncated(wal_id) => write!(f, "WAL was truncated at file {}", *wal_id), WalError::Closed => write!(f, "WAL is closed"), WalError::Unavailable(source) => write!(f, "WAL is unavailable: {source}"), WalError::DataError(source) => write!(f, "WAL data error: {source}"), @@ -113,10 +113,9 @@ impl WriterManifest { /// The result returned by [`WriterInit::fence_and_init`] pub struct WriterInitResult { - // TODO: change me to an iterator /// An iterator that returns writes that must be replayed before starting SlateDB to recover /// data from the WAL. - pub replay_range: WalFileRange, + pub replay_iterator: Box, /// The WAL writer that will be used to append new writes to the WAL pub wal_writer: Box, } @@ -230,6 +229,43 @@ pub trait WalWriter: Send { async fn close(&mut self) -> Result<(), WalError>; } +/// Rows returned by [`WalIterator`] +#[derive(Clone)] +pub struct WalRows { + /// The rows read from the WAL File. All the rows with a given sequence number must be present + /// in th same [`WalRows`]. + pub rows: Vec, + /// The id of the last WAL File for which all rows have been consumed by the iterator and + /// returned wither in this [`WalRows`] or a [`WalRows`] returned by an earlier call to + /// [`WalIterator::next`] + pub last_consumed_wal_file_id: u64, +} + +/// An iterator over rows in some range of the WAL +#[async_trait] +pub trait WalIterator: Send + 'static { + /// Returns the next set of rows. Rows must be returned in sequence and WAL File order. + /// Returns None when iterator's range is exhausted. Iterators created using an unbounded + /// end range that have exhausted the current WAL block until new rows are appended and never + /// return `None`. + /// Returns [`WalError::WalTruncated`] if the iterator observes that the WAL was truncated + /// while iterating. + async fn next(&mut self) -> Result, WalError>; +} + +/// API for reading from the WAL. Used by the Reader/ +#[async_trait] +pub trait WalReader { + /// Returns an iterator over the specified range of WAL File IDs. The start of the range must + /// not be `Unbounded`. If the end of the range is `Unbounded` then the returned iterator + /// continues returning writes as new writes are appended to the WAL. Otherwise, it returns + /// `None` upon reaching the end of the range. + async fn iterator( + &self, + wal_file_id_range: WalFileRange, + ) -> Result, WalError>; +} + impl From for WalError { fn from(status: WalStatus) -> Self { status @@ -246,18 +282,16 @@ impl From for SlateDBError { impl From for WalError { fn from(value: SlateDBError) -> Self { - { - let public: crate::Error = value.clone().into(); - match public.kind() { - ErrorKind::Closed(CloseReason::Fenced) => WalError::Fenced, - ErrorKind::Closed(CloseReason::Clean) => WalError::Closed, - ErrorKind::Closed(_) => WalError::InternalError(Arc::new(value)), - ErrorKind::Unavailable => WalError::Unavailable(Arc::new(value)), - ErrorKind::Invalid => WalError::InternalError(Arc::new(value)), - ErrorKind::Data => WalError::DataError(Arc::new(value)), - ErrorKind::Internal => WalError::InternalError(Arc::new(value)), - ErrorKind::Transaction => WalError::InternalError(Arc::new(value)), - } + let public: crate::Error = value.clone().into(); + match public.kind() { + ErrorKind::Closed(CloseReason::Fenced) => WalError::Fenced, + ErrorKind::Closed(CloseReason::Clean) => WalError::Closed, + ErrorKind::Closed(_) => WalError::InternalError(Arc::new(value)), + ErrorKind::Unavailable => WalError::Unavailable(Arc::new(value)), + ErrorKind::Invalid => WalError::InternalError(Arc::new(value)), + ErrorKind::Data => WalError::DataError(Arc::new(value)), + ErrorKind::Internal => WalError::InternalError(Arc::new(value)), + ErrorKind::Transaction => WalError::InternalError(Arc::new(value)), } } } @@ -266,7 +300,7 @@ impl From for SlateDBError { fn from(value: WalError) -> Self { match value { WalError::Fenced => SlateDBError::Fenced, - WalError::WalTruncated => SlateDBError::WalTruncated, + WalError::WalTruncated(wal_id) => SlateDBError::WalTruncated(wal_id), WalError::Closed => SlateDBError::Closed, WalError::Unavailable(err) => SlateDBError::WalUnavailable(err), WalError::DataError(err) => SlateDBError::WalDataError(err), @@ -288,7 +322,10 @@ mod tests { }; assert_eq!(WalError::Fenced.to_string(), "WAL writer was fenced"); - assert_eq!(WalError::WalTruncated.to_string(), "WAL was truncated"); + assert_eq!( + WalError::WalTruncated(123).to_string(), + "WAL was truncated at file 123" + ); assert_eq!(WalError::Closed.to_string(), "WAL is closed"); assert_eq!( WalError::Unavailable(source()).to_string(), diff --git a/slatedb/src/wal/writer_init.rs b/slatedb/src/wal/writer_init.rs index f98c85ff42..dac03c3968 100644 --- a/slatedb/src/wal/writer_init.rs +++ b/slatedb/src/wal/writer_init.rs @@ -1,10 +1,13 @@ use crate::dispatcher::MessageHandlerExecutor; use crate::error::SlateDBError; +use crate::iter::IterationOrder; use crate::manifest::Manifest; +use crate::sst_iter::SstIteratorOptions; use crate::tablestore::TableStore; use crate::utils::WatchableOnceCellReader; use crate::wal::{WalError, WriterInitResult, WriterManifest}; use crate::wal_buffer::WalBufferManager; +use crate::wal_replay::{WalIterator, WalIteratorOptions}; use crate::{wal, Settings}; use async_trait::async_trait; use fail_parallel::{fail_point_send, FailPointTx}; @@ -109,6 +112,23 @@ impl wal::WriterInit for WalWriterInit { // older writers would have failed with a stale epoch let replay_after_wal_id = manifest.core().replay_after_wal_id; assert!(empty_wal_id > replay_after_wal_id); + let replay_iterator = WalIterator::range( + replay_after_wal_id + 1..empty_wal_id + 1, + WalIteratorOptions { + sst_batch_size: 4, + sst_iter_options: SstIteratorOptions { + max_fetch_tasks: 1, + blocks_to_fetch: 256, + cache_blocks: false, + cache_metadata: false, + eager_spawn: true, + order: IterationOrder::Ascending, + prefix: None, + filter_context: None, + }, + }, + self.table_store.clone(), + )?; let wal_writer = WalBufferManager::start_new( self.closed_result_reader.clone(), &self.recorder, @@ -120,7 +140,7 @@ impl wal::WriterInit for WalWriterInit { ) .await?; let result = WriterInitResult { - replay_range: (replay_after_wal_id + 1..empty_wal_id + 1).into(), + replay_iterator: Box::new(replay_iterator), wal_writer: Box::new(wal_writer), }; return Ok(result); diff --git a/slatedb/src/wal_buffer.rs b/slatedb/src/wal_buffer.rs index a200208080..543eb79b3a 100644 --- a/slatedb/src/wal_buffer.rs +++ b/slatedb/src/wal_buffer.rs @@ -376,7 +376,7 @@ impl WalBufferManagerInner { self.last_flushed_wal_id = flushed_wal_id; if let Some(seq) = flushed_wal.last_seq() { if let Some(last_flushed_seq) = self.last_flushed_seq { - assert!(seq >= last_flushed_seq); + assert!(seq > last_flushed_seq); } self.last_flushed_seq = Some(seq); } diff --git a/slatedb/src/wal_replay.rs b/slatedb/src/wal_replay.rs index c98277ab98..1fe70eaafc 100644 --- a/slatedb/src/wal_replay.rs +++ b/slatedb/src/wal_replay.rs @@ -7,6 +7,9 @@ use crate::mem_table::WritableKVTable; use crate::sst_iter::{SstIterator, SstIteratorOptions}; use crate::tablestore::TableStore; use crate::utils::panic_string; +use crate::wal::{WalError, WalIterator as WalIteratorTrait, WalRows}; +use crate::RowEntry; +use async_trait::async_trait; use log::error; use std::collections::VecDeque; use std::ops::Range; @@ -14,17 +17,28 @@ use std::sync::Arc; use tokio::task; use tokio::task::JoinHandle; -pub(crate) struct WalReplayOptions { +pub(crate) struct WalIteratorOptions { /// The number of SSTs to preload while replaying pub(crate) sst_batch_size: usize, - /// The target maximum number of bytes in each returned table. WAL replay only - /// splits between complete WAL SSTs, so a returned table may exceed this if a - /// single WAL SST is larger. - pub(crate) max_memtable_bytes: usize, - /// Options to pass through to underlying SST iterators pub(crate) sst_iter_options: SstIteratorOptions, +} + +impl Default for WalIteratorOptions { + fn default() -> Self { + Self { + sst_batch_size: 4, + sst_iter_options: SstIteratorOptions::default(), + } + } +} + +pub(crate) struct WalReplayOptions { + /// The target maximum number of bytes in each returned table. WAL replay only + /// splits between write batches (all rows of one commit stay in one table), so + /// a returned table may exceed this if a single write batch is larger. + pub(crate) max_memtable_bytes: usize, /// The minimum seq number to replay. If unset, will replay all /// entries after `last_l0_seq` in the manifest. @@ -34,9 +48,7 @@ pub(crate) struct WalReplayOptions { impl Default for WalReplayOptions { fn default() -> Self { Self { - sst_batch_size: 4, max_memtable_bytes: 64 * 1024 * 1024, - sst_iter_options: SstIteratorOptions::default(), min_seq: None, } } @@ -49,96 +61,283 @@ pub(crate) struct ReplayedMemtable { pub(crate) last_wal_id: u64, } -struct WalIdAndIter { +pub(crate) struct WalReplayIterator { + options: WalReplayOptions, + table_store: Arc, + wal_iter: Box, + terminal_result: Option>, + /// The greatest WAL ID such that it and every WAL file before it in the replay + /// range are fully applied to returned tables. Tables are tagged with this + /// conservative watermark so that a table ending mid-file never claims a WAL + /// file it only partially contains. + last_consumed_wal_file_id: u64, + last_tick: i64, + last_seq: u64, + min_seq: u64, +} + +impl WalReplayIterator { + pub(crate) fn range( + wal_id_range: Range, + db_state: &ManifestCore, + iterator_options: WalIteratorOptions, + replay_options: WalReplayOptions, + table_store: Arc, + ) -> Result { + let wal_iter = + WalIterator::range(wal_id_range, iterator_options, Arc::clone(&table_store))?; + Self::for_wal_iterator(Box::new(wal_iter), db_state, replay_options, table_store) + } + + pub(crate) fn for_wal_iterator( + wal_iter: Box, + db_state: &ManifestCore, + options: WalReplayOptions, + table_store: Arc, + ) -> Result { + // load the last seq number from manifest, and use it as the starting seq number to avoid + // replaying the entries that are already in the L0 SST. while replaying the WALs, we'll + // update the last seq number to the max seq number, and this final `last_seq` will be passed + // to the db_state for the further writes. + let min_seq = options.min_seq.unwrap_or(db_state.last_l0_seq); + let last_seq = db_state.last_l0_seq; + let last_tick = db_state.last_l0_clock_tick; + + Ok(WalReplayIterator { + options, + table_store, + wal_iter, + terminal_result: None, + last_consumed_wal_file_id: db_state.replay_after_wal_id, + last_tick, + last_seq, + min_seq, + }) + } + + /// Get the next table replayed from the WAL. Replay accumulates write batches + /// until the returned table reaches [`WalReplayOptions::max_memtable_bytes`]. + /// Tables are only split between write batches — all rows sharing a commit seq + /// stay in one table, and batches are applied in ascending seq order — so a + /// returned table may exceed the target when a single write batch is larger. + /// + /// The returned table's `last_wal_id` is a conservative watermark: the greatest + /// WAL ID such that it and every WAL file before it are fully contained in the + /// tables returned so far. A table that ends mid-file is tagged with the last + /// fully replayed WAL ID, so replaying from `last_wal_id + 1` and dropping rows + /// with seq <= the table's `last_seq` never misses or duplicates a commit. + pub(crate) async fn next(&mut self) -> Result, SlateDBError> { + if let Some(terminal_result) = self.terminal_result.clone() { + return terminal_result.map(|_v| None); + } + + let table = WritableKVTable::new(); + let mut applied_any = false; + + loop { + let writes = match self.wal_iter.next().await { + Ok(Some(writes)) => writes, + Ok(None) => { + // we've reached the end of iteration, mark the iterator as done + self.terminal_result = Some(Ok(())); + break; + } + // Hold the error back so the write batches already applied to this + // table are returned first. `DbReader` treats a missing WAL file as + // the end of the WAL, so rows replayed before the error must not be + // dropped with it. + Err(err) => { + self.terminal_result = Some(Err(err.into())); + break; + } + }; + + applied_any = true; + assert!( + writes.last_consumed_wal_file_id >= self.last_consumed_wal_file_id, + "WAL iterator moved its consumed file watermark backwards" + ); + self.last_consumed_wal_file_id = writes.last_consumed_wal_file_id; + + for row_entry in writes.rows { + // skip the entries that are already in the L0 SST. + if row_entry.seq <= self.min_seq { + continue; + } + + if let Some(ts) = row_entry.create_ts { + self.last_tick = self.last_tick.max(ts); + } + self.last_seq = self.last_seq.max(row_entry.seq); + table.put(row_entry); + } + + if !table.is_empty() { + let meta = table.metadata(); + let estimated_bytes = self + .table_store + .estimate_encoded_size_compacted(meta.entry_num, meta.entries_size_in_bytes); + if estimated_bytes >= self.options.max_memtable_bytes { + break; + } + } + } + + if applied_any { + // we use the applied_any check here rather than checking non-empty table size to + // ensure that if we replayed values with a lower seq num we still carry the + // wal id in an empty replayed memtable + Ok(Some(ReplayedMemtable { + table, + last_tick: self.last_tick, + last_seq: self.last_seq, + last_wal_id: self.last_consumed_wal_file_id, + })) + } else { + self.terminal_result + .clone() + .expect("applied_any false but no terminal result") + .map(|_v| None) + } + } +} + +struct WalRowsCollector { wal_id: u64, iter: Box, + rows: Vec, + drained: bool, } -struct IteratorHolder { +impl WalRowsCollector { + fn new(wal_id: u64, iter: Box) -> Self { + Self { + wal_id, + iter, + rows: vec![], + drained: false, + } + } + + async fn collect(&mut self) -> Result<(), WalError> { + loop { + match self.iter.next().await { + Ok(Some(row)) => self.rows.push(row), + Ok(None) => { + self.drained = true; + break Ok(()); + } + Err(err) if err.has_object_store_not_found() => { + break Err(WalError::WalTruncated(self.wal_id)); + } + Err(err) => { + break Err(err.into()); + } + } + } + } +} + +impl From for WalRows { + fn from(reader: WalRowsCollector) -> Self { + assert!(reader.drained); + WalRows { + last_consumed_wal_file_id: reader.wal_id, + rows: reader.rows, + } + } +} + +struct CurrentWalFile { initialized: bool, - current_iter: Option, + collector: Option, } -impl IteratorHolder { - fn new() -> Self { +impl CurrentWalFile { + fn initial() -> Self { Self { initialized: false, - current_iter: None, + collector: None, } } - fn is_finished(&self) -> bool { - self.initialized && self.current_iter.is_none() + fn initialized(&self) -> bool { + self.initialized + } + + async fn collect(&mut self) -> Result, WalError> { + assert!(self.initialized); + let Some(collector) = &mut self.collector else { + return Ok(None); + }; + collector.collect().await?; + let collector = self.collector.take().expect("unreachable"); + self.initialized = false; + Ok(Some(collector.into())) } - fn advance(&mut self, iterator: Option) { + fn advance(&mut self, collector: WalRowsCollector) { + assert!(!self.initialized); self.initialized = true; - self.current_iter = iterator; + self.collector = Some(collector); } - fn reset(&mut self) { - self.initialized = false; - self.current_iter = None; + fn finish(&mut self) { + self.initialized = true; + self.collector = None; } } -pub(crate) struct WalReplayIterator { - options: WalReplayOptions, +/// Iterates over the writes in a range of WAL files, preloading up to +/// `sst_batch_size` WAL SSTs concurrently. Returns the rows of one WAL file per +/// [`WalRows`], and verifies that files carry strictly increasing seq +/// ranges — the ordering callers rely on to split and tag memtables safely. +/// +/// Preloading only opens each WAL SST (footer, index, and any eagerly fetched +/// blocks); a file's rows are read out only when it is returned from +/// [`Self::next`], so at most one file's rows are materialized at a time. +pub(crate) struct WalIterator { + options: WalIteratorOptions, + /// Range of WAL IDs to iterate over wal_id_range: Range, table_store: Arc, - current_iter: IteratorHolder, - next_iters: VecDeque, SlateDBError>>>, - last_tick: i64, - last_seq: u64, - min_seq: u64, + next_files: VecDeque>>, next_wal_id: u64, + /// The greatest seq returned so far, used to verify that WAL files arrive + /// with strictly increasing seq ranges. + last_seq: Option, + /// Set once iteration has ended, either because the range was exhausted or + /// because an error was returned. + terminal_result: Option, WalError>>, + current_file: CurrentWalFile, } -impl WalReplayIterator { - pub(crate) async fn range( +impl WalIterator { + pub(crate) fn range( wal_id_range: Range, - db_state: &ManifestCore, - options: WalReplayOptions, + options: WalIteratorOptions, table_store: Arc, ) -> Result { - let sst_batch_size = options.sst_batch_size; - if sst_batch_size < 1 { - return Err(SlateDBError::InvalidSSTBatchSize(sst_batch_size)); + if options.sst_batch_size < 1 { + return Err(SlateDBError::InvalidSSTBatchSize(options.sst_batch_size)); } - // load the last seq number from manifest, and use it as the starting seq number to avoid - // replaying the entries that are already in the L0 SST. while replaying the WALs, we'll - // update the last seq number to the max seq number, and this final `last_seq` will be passed - // to the db_state for the further writes. - let min_seq = options.min_seq.unwrap_or(db_state.last_l0_seq); - let last_seq = db_state.last_l0_seq; - let last_tick = db_state.last_l0_clock_tick; let next_wal_id = wal_id_range.start; - - let mut replay_iter = WalReplayIterator { + Ok(WalIterator { options, wal_id_range, - table_store: Arc::clone(&table_store), - current_iter: IteratorHolder::new(), - next_iters: VecDeque::new(), - last_tick, - last_seq, - min_seq, + table_store, + next_files: VecDeque::new(), next_wal_id, - }; - - for _ in 0..sst_batch_size { - if !replay_iter.maybe_load_next_iter() { - break; - } - } - - Ok(replay_iter) + last_seq: None, + terminal_result: None, + current_file: CurrentWalFile::initial(), + }) } - fn maybe_load_next_iter(&mut self) -> bool { + fn maybe_spawn_open(&mut self) -> bool { if !self.wal_id_range.contains(&self.next_wal_id) - || self.next_iters.len() >= self.options.sst_batch_size + || self.next_files.len() >= self.options.sst_batch_size { return false; } @@ -146,20 +345,20 @@ impl WalReplayIterator { let next_wal_id = self.next_wal_id; self.next_wal_id += 1; - async fn load_iter( + async fn try_open_file_iter( wal_id: u64, sst_iter_options: SstIteratorOptions, table_store: Arc, - ) -> Result, SlateDBError> { + ) -> Result { let sst = match table_store.open_sst(&SsTableId::Wal(wal_id)).await { Ok(sst) => sst, Err(SlateDBError::EmptySSTable) => { // Zero-byte WAL files are fence markers; replay them as empty WALs // so the last replayed WAL ID still advances past the marker. - return Ok(Some(WalIdAndIter { + return Ok(WalRowsCollector::new( wal_id, - iter: Box::new(EmptyIterator::new()), - })); + Box::new(EmptyIterator::new()), + )); } Err(err) => return Err(err), }; @@ -170,111 +369,145 @@ impl WalReplayIterator { sst_iter_options, ) .await?; - Ok(iter.map(|iter| WalIdAndIter { - wal_id, - iter: Box::new(iter) as Box, - })) + // An unbounded, unfiltered scan over a WAL SST always yields an + // iterator. `None` means the file cannot be read, and replay must + // fail rather than silently end early and drop the remaining WALs. + let Some(iter) = iter else { + error!( + "could not construct row iterator over WAL SST. [wal_id={}]", + wal_id + ); + return Err(SlateDBError::InvalidDBState); + }; + Ok(WalRowsCollector::new(wal_id, Box::new(iter))) + } + + async fn open_file_iter( + wal_id: u64, + sst_iter_options: SstIteratorOptions, + table_store: Arc, + ) -> Result { + match try_open_file_iter(wal_id, sst_iter_options, table_store).await { + Ok(iter) => Ok(iter), + Err(err) if err.has_object_store_not_found() => Err(WalError::WalTruncated(wal_id)), + Err(err) => Err(err.into()), + } } - let handle = task::spawn(load_iter( + let handle = task::spawn(open_file_iter( next_wal_id, self.options.sst_iter_options.clone(), Arc::clone(&self.table_store), )); - self.next_iters.push_back(handle); + self.next_files.push_back(handle); true } - async fn advance_current_iter(&mut self) -> Result<(), SlateDBError> { - let next_iter = if let Some(join_handle) = self.next_iters.pop_front() { - match join_handle.await { - Ok(Ok(sst_iter)) => sst_iter, - Ok(Err(slate_err)) => return Err(slate_err), - Err(join_err) => { - let task_name = format!("wal_replay[{:?}]", self.wal_id_range); - if let Ok(panic_err) = join_err.try_into_panic() { - error!( - "wal_replay task panicked unexpectedly. [task_name={}, panic={}]", - task_name, - panic_string(&panic_err), - ); - return Err(SlateDBError::BackgroundTaskPanic(task_name)); - } - return Err(SlateDBError::BackgroundTaskCancelled(task_name)); - } - } - } else { - None + /// Await the next preloaded WAL file and return an iterator over its rows. + /// Returns `None` when there are no more files to read. + async fn load_next_file(&mut self) -> Result<(), WalError> { + if self.current_file.initialized() { + return Ok(()); + } + // await a mutable ref to the task so that next remains cancel-safe + // see https://docs.rs/tokio/latest/tokio/task/struct.JoinHandle.html#cancel-safety + let Some(join_handle) = self.next_files.front_mut() else { + self.current_file.finish(); + return Ok(()); }; - self.current_iter.advance(next_iter); - Ok(()) + let result = join_handle.await; + self.next_files.pop_front(); + match result { + Ok(result) => { + self.current_file.advance(result?); + Ok(()) + } + Err(join_err) => { + let task_name = format!("wal_replay[{:?}]", self.wal_id_range); + let msg = if let Ok(panic_err) = join_err.try_into_panic() { + format!( + "wal_replay task panicked unexpectedly. [task_name={}, panic={}]", + task_name, + panic_string(&panic_err), + ) + } else { + format!("wal_replay task cancelled. [task_name={}]", task_name) + }; + error!("{}", msg); + let error = Arc::from(Box::::from(msg)); + Err(WalError::InternalError(error)) + } + } } - /// Get the next table replayed from the WAL. Replay accumulates complete WAL - /// SSTs until the returned table reaches [`WalReplayOptions::max_memtable_bytes`], - /// unless it is the final table replayed from the WAL. The final table may even - /// be empty since writers use an empty WAL to fence zombie writers. The empty - /// table must still be returned so that replay logic can account for the latest - /// WAL ID. - /// - /// The returned table may exceed [`WalReplayOptions::max_memtable_bytes`] when - /// a complete WAL SST is larger than the configured target, because replay - /// must not split a WAL SST across replayed memtables. - pub(crate) async fn next(&mut self) -> Result, SlateDBError> { - if self.current_iter.is_finished() { - return Ok(None); + fn terminate( + &mut self, + result: Result, WalError>, + ) -> Result, WalError> { + self.terminal_result = Some(result.clone()); + for task in self.next_files.drain(..) { + task.abort(); } + result + } +} - let table = WritableKVTable::new(); - let mut last_wal_id = 0; - - while !self.current_iter.is_finished() { - if let Some(wal_id_and_iter) = &mut self.current_iter.current_iter { - while let Some(row_entry) = wal_id_and_iter.iter.next().await? { - // skip the entries that are already in the L0 SST. - if row_entry.seq <= self.min_seq { - continue; - } +#[async_trait] +impl WalIteratorTrait for WalIterator { + /// Get the next set of writes from the WAL files in the range. Each returned + /// [`WalRows`] holds the rows of one WAL file; a WAL file with no rows + /// yields a batch with empty `rows`. Returns `None` once all WAL files in the + /// range have been read. It is an error if a WAL file in the range is not + /// present. Errors are returned only on calls that return no batch, so rows + /// read from earlier WAL files are never dropped with a later file's error. + async fn next(&mut self) -> Result, WalError> { + if let Some(result) = self.terminal_result.clone() { + return result; + } - if let Some(ts) = row_entry.create_ts { - self.last_tick = self.last_tick.max(ts); + while self.maybe_spawn_open() {} + if let Err(err) = self.load_next_file().await { + return self.terminate(Err(err)); + } + match self.current_file.collect().await { + Err(err) => self.terminate(Err(err)), + Ok(None) => self.terminate(Ok(None)), + Ok(Some(rows)) => { + // Verify that WAL files carry strictly increasing seq ranges. Replay + // relies on this ordering to split and tag memtables safely: a commit seq + // spanning two WAL files, or files with overlapping seq ranges, would + // break recovery's (wal_id, seq) watermark filtering. + if let Some(min_seq) = rows.rows.iter().map(|row| row.seq).min() { + if let Some(last_seq) = self.last_seq { + if min_seq <= last_seq { + let msg = format!( + "WAL replay saw out-of-order seqs across WAL files. \ + [wal_id={}, min_seq={}, last_seq={}]", + rows.last_consumed_wal_file_id, min_seq, last_seq, + ); + error!("{}", &msg); + let error = + Arc::from(Box::::from(msg)); + return self.terminate(Err(WalError::InternalError(error))); + } } - self.last_seq = self.last_seq.max(row_entry.seq); - table.put(row_entry); - } - - last_wal_id = wal_id_and_iter.wal_id; - - let meta = table.metadata(); - let estimated_bytes = self - .table_store - .estimate_encoded_size_compacted(meta.entry_num, meta.entries_size_in_bytes); - if !table.is_empty() && estimated_bytes >= self.options.max_memtable_bytes { - self.current_iter.reset(); - break; + let max_seq = rows + .rows + .iter() + .map(|row| row.seq) + .max() + .expect("non-empty rows have a max seq"); + self.last_seq = Some(max_seq); } + Ok(Some(rows)) } - - self.maybe_load_next_iter(); - self.advance_current_iter().await? - } - - if last_wal_id > 0 { - Ok(Some(ReplayedMemtable { - table, - last_tick: self.last_tick, - last_seq: self.last_seq, - last_wal_id, - })) - } else { - Ok(None) } } } #[cfg(test)] mod tests { - use super::{WalReplayIterator, WalReplayOptions}; + use super::{WalIterator, WalIteratorOptions, WalReplayIterator, WalReplayOptions}; use crate::block_cache_policy::BlockCachePolicy; use crate::bytes_range::BytesRange; use crate::db_state::SsTableId; @@ -286,7 +519,9 @@ mod tests { use crate::proptest_util::{rng, sample}; use crate::tablestore::{TableStore, TableStoreKind}; use crate::types::RowEntry; + use crate::wal::{WalError, WalIterator as WalIteratorTrait, WalRows}; use crate::{error::SlateDBError, test_utils}; + use async_trait::async_trait; use bytes::Bytes; use object_store::memory::InMemory; use object_store::path::Path; @@ -295,9 +530,20 @@ mod tests { use rand::Rng; use std::cmp::min; use std::collections::btree_map::Iter; - use std::collections::BTreeMap; + use std::collections::{BTreeMap, BTreeSet, VecDeque}; use std::sync::Arc; + struct ScriptedWalIterator { + results: VecDeque, WalError>>, + } + + #[async_trait] + impl WalIteratorTrait for ScriptedWalIterator { + async fn next(&mut self) -> Result, WalError> { + self.results.pop_front().unwrap_or(Ok(None)) + } + } + impl WalReplayIterator { async fn all_wal_ids( db_state: &ManifestCore, @@ -309,10 +555,164 @@ mod tests { .last_seen_wal_id(db_state.replay_after_wal_id) .await?; let wal_id_range = wal_id_start..(wal_id_end + 1); - Self::range(wal_id_range, db_state, options, table_store).await + Self::range( + wal_id_range, + db_state, + WalIteratorOptions::default(), + options, + table_store, + ) } } + #[tokio::test] + async fn should_return_replayed_rows_before_repeating_terminal_error() { + let table_store = test_table_store(); + let first_row = RowEntry::new_value(b"key_001", b"value_001", 1); + let later_row = RowEntry::new_value(b"key_002", b"value_002", 2); + let wal_iter = ScriptedWalIterator { + results: VecDeque::from([ + Ok(Some(WalRows { + rows: vec![first_row], + last_consumed_wal_file_id: 1, + })), + Err(WalError::WalTruncated(2)), + // A terminal error must prevent the replay iterator from resuming + // the underlying iterator on later calls. + Ok(Some(WalRows { + rows: vec![later_row], + last_consumed_wal_file_id: 3, + })), + ]), + }; + let mut replay_iter = WalReplayIterator::for_wal_iterator( + Box::new(wal_iter), + &ManifestCore::new(), + WalReplayOptions { + max_memtable_bytes: usize::MAX, + ..WalReplayOptions::default() + }, + Arc::clone(&table_store), + ) + .unwrap(); + + let replayed = replay_iter.next().await.unwrap().unwrap(); + assert_eq!(replayed.last_wal_id, 1); + assert_eq!(replayed.last_seq, 1); + assert_eq!(replayed.table.metadata().entry_num, 1); + + assert!(matches!( + replay_iter.next().await, + Err(SlateDBError::WalTruncated(2)) + )); + assert!(matches!( + replay_iter.next().await, + Err(SlateDBError::WalTruncated(2)) + )); + } + + #[tokio::test] + async fn should_repeat_terminal_none_for_wal_replay_iterator() { + let table_store = test_table_store(); + let wal_iter = ScriptedWalIterator { + results: VecDeque::from([ + Ok(None), + // Normal termination must prevent the replay iterator from + // resuming the underlying iterator on later calls. + Ok(Some(WalRows { + rows: vec![RowEntry::new_value(b"key", b"value", 1)], + last_consumed_wal_file_id: 1, + })), + ]), + }; + let mut replay_iter = WalReplayIterator::for_wal_iterator( + Box::new(wal_iter), + &ManifestCore::new(), + WalReplayOptions::default(), + Arc::clone(&table_store), + ) + .unwrap(); + + assert!(replay_iter.next().await.unwrap().is_none()); + assert!(replay_iter.next().await.unwrap().is_none()); + } + + #[tokio::test] + async fn should_repeat_terminal_error_for_wal_iterator() { + let table_store = test_table_store(); + let mut wal_iter = WalIterator::range( + 1..2, + WalIteratorOptions::default(), + Arc::clone(&table_store), + ) + .unwrap(); + + assert!(matches!( + wal_iter.next().await, + Err(WalError::WalTruncated(1)) + )); + assert!(matches!( + wal_iter.next().await, + Err(WalError::WalTruncated(1)) + )); + } + + #[tokio::test] + async fn should_repeat_terminal_none_for_wal_iterator() { + let table_store = test_table_store(); + let mut wal_iter = WalIterator::range( + 1..1, + WalIteratorOptions::default(), + Arc::clone(&table_store), + ) + .unwrap(); + + assert!(wal_iter.next().await.unwrap().is_none()); + assert!(wal_iter.next().await.unwrap().is_none()); + } + + #[tokio::test] + async fn should_use_last_consumed_wal_file_id_as_replay_watermark() { + let table_store = test_table_store(); + let first_row = RowEntry::new_value(b"key_001", &[b'x'; 128], 1); + let second_row = RowEntry::new_value(b"key_002", &[b'x'; 128], 2); + let max_memtable_bytes = + table_store.estimate_encoded_size_compacted(1, first_row.estimated_size()); + let wal_iter = ScriptedWalIterator { + results: VecDeque::from([ + Ok(Some(WalRows { + rows: vec![first_row], + // The first batch ends partway through WAL file 1. + last_consumed_wal_file_id: 0, + })), + Ok(Some(WalRows { + rows: vec![second_row], + // The second batch consumes the rest of WAL file 1. + last_consumed_wal_file_id: 1, + })), + ]), + }; + let mut replay_iter = WalReplayIterator::for_wal_iterator( + Box::new(wal_iter), + &ManifestCore::new(), + WalReplayOptions { + max_memtable_bytes, + ..WalReplayOptions::default() + }, + Arc::clone(&table_store), + ) + .unwrap(); + + let first = replay_iter.next().await.unwrap().unwrap(); + assert_eq!(first.last_wal_id, 0); + assert_eq!(first.last_seq, 1); + + let second = replay_iter.next().await.unwrap().unwrap(); + assert_eq!(second.last_wal_id, 1); + assert_eq!(second.last_seq, 2); + assert!(replay_iter.next().await.unwrap().is_none()); + } + #[tokio::test] async fn should_replay_empty_wal() { let table_store = test_table_store(); @@ -661,6 +1061,115 @@ mod tests { } } + #[tokio::test] + async fn should_return_atomic_wal_rows_in_increasing_seq_order() { + let table_store = test_table_store(); + // Each file contains out-of-order rows and a sequence that appears twice. + // The iterator must keep both rows for a sequence in one batch, while the + // sequence range of the second batch must follow the first. + let wal_entries = [ + vec![ + RowEntry::new_value(b"key_001", &[b'x'; 128], 2), + RowEntry::new_value(b"key_002", &[b'x'; 128], 1), + RowEntry::new_value(b"key_003", &[b'x'; 128], 2), + ], + vec![ + RowEntry::new_value(b"key_004", &[b'x'; 128], 4), + RowEntry::new_value(b"key_005", &[b'x'; 128], 3), + RowEntry::new_value(b"key_006", &[b'x'; 128], 4), + ], + ]; + let mut expected_rows = BTreeMap::new(); + let mut expected_rows_by_seq = BTreeMap::>::new(); + let wal_file_count = wal_entries.len() as u64; + for (file_index, entries) in wal_entries.iter().enumerate() { + let wal_id = file_index as u64 + 1; + for row in entries { + expected_rows.insert(row.key.clone(), (row.clone(), wal_id)); + expected_rows_by_seq + .entry(row.seq) + .or_default() + .insert(row.key.clone()); + } + } + for (index, entries) in wal_entries.into_iter().enumerate() { + let mut builder = table_store.wal_table_builder(); + for entry in entries { + builder.add(entry).await.unwrap(); + } + let encoded_sst = builder.build().await.unwrap(); + table_store + .write_sst(&SsTableId::Wal(index as u64 + 1), &encoded_sst) + .await + .unwrap(); + } + let mut wal_iter = WalIterator::range( + 1..(wal_file_count + 1), + WalIteratorOptions::default(), + Arc::clone(&table_store), + ) + .unwrap(); + + let mut returned_rows = BTreeMap::new(); + let mut previous_max_seq = None; + let mut last_consumed_wal_file_id = 0; + while let Some(batch) = wal_iter.next().await.unwrap() { + let batch_min_seq = batch.rows.iter().map(|r| r.seq).min().unwrap(); + let batch_max_seq = batch.rows.iter().map(|r| r.seq).max().unwrap(); + if let Some(previous_max_seq) = previous_max_seq { + assert!( + batch_min_seq > previous_max_seq, + "consecutive WAL batches have overlapping sequence ranges" + ); + } + previous_max_seq = Some(batch_max_seq); + + let mut batch_rows_by_seq = BTreeMap::>::new(); + for row in &batch.rows { + assert!( + returned_rows.insert(row.key.clone(), row.clone()).is_none(), + "row was returned more than once: {:?}", + row.key + ); + batch_rows_by_seq + .entry(row.seq) + .or_default() + .insert(row.key.clone()); + } + for (seq, batch_rows) in batch_rows_by_seq { + assert_eq!( + expected_rows_by_seq.get(&seq), + Some(&batch_rows), + "rows for seq {seq} were split across WAL batches" + ); + } + + assert!( + batch.last_consumed_wal_file_id >= last_consumed_wal_file_id, + "consumed WAL file watermark moved backwards" + ); + assert!(batch.last_consumed_wal_file_id <= wal_file_count); + for wal_id in 1..=batch.last_consumed_wal_file_id { + let file_fully_returned = + expected_rows.iter().all(|(key, (_, expected_wal_id))| { + *expected_wal_id != wal_id || returned_rows.contains_key(key) + }); + assert!( + file_fully_returned, + "WAL file {wal_id} was marked consumed before all its rows were returned" + ); + } + last_consumed_wal_file_id = batch.last_consumed_wal_file_id; + } + + let expected_returned_rows = expected_rows + .into_iter() + .map(|(key, (row, _wal_id))| (key, row)) + .collect(); + assert_eq!(returned_rows, expected_returned_rows); + assert_eq!(last_consumed_wal_file_id, wal_file_count); + } + #[tokio::test] async fn should_only_replay_wals_after_last_l0_flushed_wal_id() { let table_store = test_table_store(); From f73073542027c2b473748aed9ea5fc5cd3ea6ce9 Mon Sep 17 00:00:00 2001 From: Rohan Date: Wed, 29 Jul 2026 20:01:40 -0400 Subject: [PATCH 03/65] disallow clone with projection when WAL must be copied (#1984) --- slatedb/src/clone.rs | 201 ++++++++++++++++++++++++++++++++++++++++--- slatedb/src/error.rs | 6 +- 2 files changed, 192 insertions(+), 15 deletions(-) diff --git a/slatedb/src/clone.rs b/slatedb/src/clone.rs index 7f8a03644e..c37ce2d613 100644 --- a/slatedb/src/clone.rs +++ b/slatedb/src/clone.rs @@ -146,15 +146,27 @@ async fn create_clone_manifest + Clone>( ) .await?; + let projection_requested = projection_range.is_some() + || segment_filter.is_some() + || segment_projection.is_some() + || source_specs.iter().any(|s| s.projection_range.is_some()); + let manifest: Manifest = match &sources[..] { - // no need to call validate_no_data_wal() because for single source, - // WAL is copied by the caller (create_clone) - [single_source] => Manifest::cloned( - &single_source.manifest, - single_source.path.to_string(), - single_source.checkpoint.id, - rand, - ), + [single_source] => { + // WAL SSTs are copied to the clone verbatim and replayed in full + // when the clone is opened, so entries outside the projected + // range would leak into the clone. So we reject projections if + // there are non-fence WALs to copy. + if projection_requested { + validate_no_data_wal(&sources, &wal_object_store).await?; + } + Manifest::cloned( + &single_source.manifest, + single_source.path.to_string(), + single_source.checkpoint.id, + rand, + ) + } [..] => { validate_no_data_wal(&sources, &wal_object_store).await?; Manifest::cloned_from_union(sources, rand)? @@ -368,7 +380,7 @@ async fn validate_no_data_wal( } } if !parents_with_wal.is_empty() { - return Err(SlateDBError::InvalidUnionSourceWithWal { + return Err(SlateDBError::InvalidCloneSourceWithWal { paths: parents_with_wal, }); } @@ -1356,6 +1368,161 @@ mod tests { )); } + #[tokio::test] + async fn should_disallow_projected_clone_when_source_has_data_wal() { + // Data that only lives in the parent's WAL at the checkpoint is copied + // to the clone verbatim and replayed in full on first open, so a + // projection cannot be applied to it. Cloning with a projection must + // fail while the source still has data in its WAL, and succeed once + // that data has been flushed into L0. + let fp_registry = Arc::new(FailPointRegistry::new()); + let object_store: Arc = Arc::new(InMemory::new()); + let parent_path = Path::from("/tmp/test_parent_wal_projection"); + let clone_path = Path::from("/tmp/test_clone_wal_projection"); + + let parent_db = Db::builder(parent_path.clone(), object_store.clone()) + .with_fp_registry(fp_registry.clone()) + .build() + .await + .unwrap(); + let write_options = WriteOptions { + await_durable: false, + ..Default::default() + }; + let put_options = PutOptions::default(); + + // Keys inside and outside the projection range [aaa, bbb), flushed + // through to L0 ... + parent_db + .put_with_options(b"aaa-l0", b"v1", &put_options, &write_options) + .await + .unwrap(); + parent_db + .put_with_options(b"zzz-l0", b"v2", &put_options, &write_options) + .await + .unwrap(); + parent_db.flush().await.unwrap(); + parent_db + .flush_with_options(FlushOptions { + flush_type: FlushType::MemTable, + }) + .await + .unwrap(); + + // ... and the same shape of data made durable only in the WAL. + parent_db + .put_with_options(b"aaa-wal", b"v3", &put_options, &write_options) + .await + .unwrap(); + parent_db + .put_with_options(b"zzz-wal", b"v4", &put_options, &write_options) + .await + .unwrap(); + parent_db.flush().await.unwrap(); + + let manifest = parent_db.manifest(); + assert!( + !manifest.manifest.core.tree.l0.is_empty(), + "expected parent state to include L0 data" + ); + assert!( + manifest.manifest.core.replay_after_wal_id + 1 < manifest.manifest.core.next_wal_sst_id, + "expected parent state to retain WAL-only SSTs" + ); + + // Block L0 uploads so the WAL-only data stays in the WAL. + fail_parallel::cfg( + fp_registry.clone(), + "write-compacted-sst-io-error", + "return", + ) + .unwrap(); + // expect to fail since l0 upload is blocked + assert!(parent_db.close().await.is_err()); + fail_parallel::cfg(fp_registry.clone(), "write-compacted-sst-io-error", "off").unwrap(); + + // Cloning with a projection that keeps only keys in [aaa, bbb) must + // be rejected while the WAL-only data is still in the WAL. + let range = ( + Bound::Included(Bytes::from_static(b"aaa")), + Bound::Excluded(Bytes::from_static(b"bbb")), + ); + let err = crate::clone::create_clone( + vec![CloneSourceSpec::new(parent_path.clone())], + clone_path.clone(), + ObjectStores::new(object_store.clone(), Some(object_store.clone())), + Arc::new(FailPointRegistry::new()), + Arc::new(DefaultSystemClock::new()), + Arc::new(DbRand::default()), + Some(range.clone()), + None, + None, + ) + .await + .unwrap_err(); + assert!(matches!( + err, + SlateDBError::InvalidCloneSourceWithWal { ref paths } + if paths == &vec![parent_path.clone()] + )); + + // Reopen the parent so the WAL tail is replayed, flush it into L0, + // and close cleanly. With no data WALs left to copy the projected + // clone is allowed. + let parent_db = Db::open(parent_path.clone(), object_store.clone()) + .await + .unwrap(); + parent_db + .flush_with_options(FlushOptions { + flush_type: FlushType::MemTable, + }) + .await + .unwrap(); + parent_db.close().await.unwrap(); + + crate::clone::create_clone( + vec![CloneSourceSpec::new(parent_path.clone())], + clone_path.clone(), + ObjectStores::new(object_store.clone(), Some(object_store.clone())), + Arc::new(FailPointRegistry::new()), + Arc::new(DefaultSystemClock::new()), + Arc::new(DbRand::default()), + Some(range), + None, + None, + ) + .await + .unwrap(); + + let clone_db = Db::open(clone_path.clone(), object_store.clone()) + .await + .unwrap(); + + // L0 data respects the projection. + assert_eq!( + clone_db.get(b"aaa-l0").await.unwrap(), + Some(Bytes::from_static(b"v1")) + ); + assert_eq!( + clone_db.get(b"zzz-l0").await.unwrap(), + None, + "L0 entry outside the projection range must not be visible in the clone" + ); + + // The formerly WAL-only data was flushed into L0 before the retry, + // so it must respect the projection too. + assert_eq!( + clone_db.get(b"aaa-wal").await.unwrap(), + Some(Bytes::from_static(b"v3")) + ); + assert_eq!( + clone_db.get(b"zzz-wal").await.unwrap(), + None, + "entry outside the projection range must not be visible in the clone" + ); + clone_db.close().await.unwrap(); + } + fn segmented_table() -> BTreeMap { BTreeMap::from([ (Bytes::from_static(b"aaa-001"), Bytes::from_static(b"v1")), @@ -1382,6 +1549,10 @@ mod tests { settings: Settings, table: &BTreeMap, ) { + #[cfg(feature = "wal_disable")] + let wal_enabled = settings.wal_enabled; + #[cfg(not(feature = "wal_disable"))] + let wal_enabled = true; let db = Db::builder(path.clone(), object_store) .with_settings(settings) .with_segment_extractor(extractor) @@ -1391,6 +1562,12 @@ mod tests { // await_durable would deadlock under wal_enabled=false because the // memtable flush is gated on the explicit call below. test_utils::seed_database(&db, table, false).await.unwrap(); + if wal_enabled { + // Flush the WAL before the memtable so that `replay_after_wal_id` + // covers every data WAL; projected clones of this parent would + // otherwise be rejected. + db.flush().await.unwrap(); + } db.flush_with_options(FlushOptions { flush_type: FlushType::MemTable, }) @@ -2095,7 +2272,7 @@ mod tests { } /// A union clone whose source references a real (non-empty) data WAL above - /// `replay_after_wal_id` must FAIL with `InvalidUnionSourceWithWal`, since + /// `replay_after_wal_id` must FAIL with `InvalidCloneSourceWithWal`, since /// the union clone would silently drop that WAL data. #[cfg(feature = "wal_disable")] #[tokio::test] @@ -2144,10 +2321,10 @@ mod tests { .unwrap_err(); match err { - SlateDBError::InvalidUnionSourceWithWal { paths } => { + SlateDBError::InvalidCloneSourceWithWal { paths } => { assert!(paths.contains(&parent_path_a)); } - other => panic!("expected InvalidUnionSourceWithWal, got {other:?}"), + other => panic!("expected InvalidCloneSourceWithWal, got {other:?}"), } } diff --git a/slatedb/src/error.rs b/slatedb/src/error.rs index 1ade60c558..006b6a7a0c 100644 --- a/slatedb/src/error.rs +++ b/slatedb/src/error.rs @@ -218,8 +218,8 @@ pub(crate) enum SlateDBError { #[error("clone source paths must be unique, found duplicate: `{0}`")] DuplicatedCloneSourcePath(Path), - #[error("Manifest union of sources with WAL is not supported, source with WAL: `{paths:?}`")] - InvalidUnionSourceWithWal { paths: Vec }, + #[error("Projection and/or union with WAL is not supported, sources with WAL: `{paths:?}`")] + InvalidCloneSourceWithWal { paths: Vec }, #[error("Source manifest set must not be empty")] InvalidUnionSetEmpty(), @@ -675,7 +675,7 @@ impl From for Error { SlateDBError::SeekKeyLessThanLastReturnedKey => Error::invalid(msg), SlateDBError::IdenticalClonePaths { .. } => Error::invalid(msg), SlateDBError::DuplicatedCloneSourcePath(_) => Error::invalid(msg), - SlateDBError::InvalidUnionSourceWithWal { .. } => Error::invalid(msg), + SlateDBError::InvalidCloneSourceWithWal { .. } => Error::invalid(msg), SlateDBError::InvalidUnionSetEmpty() => Error::invalid(msg), SlateDBError::InvalidUnion(_) => Error::invalid(msg), SlateDBError::InvalidProjection { .. } => Error::invalid(msg), From b37f69b182df7e1d6bd571c993b05f910716cfc8 Mon Sep 17 00:00:00 2001 From: Chris Date: Wed, 29 Jul 2026 17:37:40 -0700 Subject: [PATCH 04/65] Clarify TTL millisecond units (#1989) --- bindings/go/uniffi/doc.go | 2 +- bindings/go/uniffi/slatedb.go | 30 +++---- bindings/go/uniffi/slatedb_test.go | 12 +-- bindings/uniffi/src/config.rs | 12 +-- bindings/uniffi/src/settings.rs | 18 ++-- rfcs/0003-timestamps-and-ttl.md | 10 +-- rfcs/0006-merge-operator.md | 12 +-- slatedb/benches/db_transaction.rs | 4 +- slatedb/benches/write_batch.rs | 4 +- slatedb/src/batch.rs | 40 ++++----- slatedb/src/batch_write.rs | 2 +- slatedb/src/compactor.rs | 22 ++--- slatedb/src/config.rs | 82 ++++++++++++------- slatedb/src/db.rs | 20 ++--- slatedb/src/db_transaction.rs | 8 +- website/src/content/docs/docs/design/time.mdx | 10 +-- 16 files changed, 154 insertions(+), 134 deletions(-) diff --git a/bindings/go/uniffi/doc.go b/bindings/go/uniffi/doc.go index a3a5daad72..4a12200279 100644 --- a/bindings/go/uniffi/doc.go +++ b/bindings/go/uniffi/doc.go @@ -107,7 +107,7 @@ // [Db.Write] or [Db.WriteWithOptions]. Batches are single-use once submitted. // // TTL behavior is configured with [Ttl] implementations such as [TtlDefault], -// [TtlNoExpiry], and [TtlExpireAfterTicks]. +// [TtlNoExpiry], and [TtlExpireAfterMillis]. // // # Transactions // diff --git a/bindings/go/uniffi/slatedb.go b/bindings/go/uniffi/slatedb.go index afa3390e8d..31daffe0be 100644 --- a/bindings/go/uniffi/slatedb.go +++ b/bindings/go/uniffi/slatedb.go @@ -1573,7 +1573,7 @@ func uniffiCheckChecksums() { checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { return C.uniffi_slatedb_uniffi_checksum_method_settings_set() }) - if checksum != 34344 { + if checksum != 16989 { // If this happens try cleaning and rebuilding your project panic("slatedb: uniffi_slatedb_uniffi_checksum_method_settings_set: UniFFI API checksum mismatch") } @@ -7999,8 +7999,8 @@ type SettingsInterface interface { // Examples: // // - `set("flush_interval", "\"250ms\"")` - // - `set("default_ttl", "42")` - // - `set("default_ttl", "null")` + // - `set("default_ttl_millis", "42")` + // - `set("default_ttl_millis", "null")` // - `set("compactor_options.max_sst_size", "33554432")` // - `set("object_store_cache_options.root_folder", "\"/tmp/slatedb-cache\"")` Set(key string, valueJson string) error @@ -8104,8 +8104,8 @@ func SettingsLoad() (*Settings, error) { // Examples: // // - `set("flush_interval", "\"250ms\"")` -// - `set("default_ttl", "42")` -// - `set("default_ttl", "null")` +// - `set("default_ttl_millis", "42")` +// - `set("default_ttl_millis", "null")` // - `set("compactor_options.max_sst_size", "33554432")` // - `set("object_store_cache_options.root_folder", "\"/tmp/slatedb-cache\"")` func (_self *Settings) Set(key string, valueJson string) error { @@ -12714,21 +12714,21 @@ type TtlNoExpiry struct { func (e TtlNoExpiry) Destroy() { } -// Expire the value after the given number of clock ticks. -type TtlExpireAfterTicks struct { +// Expire the value after the given number of milliseconds. +type TtlExpireAfterMillis struct { Field0 uint64 } -func (e TtlExpireAfterTicks) Destroy() { +func (e TtlExpireAfterMillis) Destroy() { FfiDestroyerUint64{}.Destroy(e.Field0) } -// Expire the value at the given absolute timestamp (clock ticks). -type TtlExpireAt struct { +// Expire the value at the given Unix timestamp in milliseconds. +type TtlExpireAtMillis struct { Field0 int64 } -func (e TtlExpireAt) Destroy() { +func (e TtlExpireAtMillis) Destroy() { FfiDestroyerInt64{}.Destroy(e.Field0) } @@ -12755,11 +12755,11 @@ func (FfiConverterTtl) Read(reader io.Reader) Ttl { case 2: return TtlNoExpiry{} case 3: - return TtlExpireAfterTicks{ + return TtlExpireAfterMillis{ FfiConverterUint64INSTANCE.Read(reader), } case 4: - return TtlExpireAt{ + return TtlExpireAtMillis{ FfiConverterInt64INSTANCE.Read(reader), } default: @@ -12773,10 +12773,10 @@ func (FfiConverterTtl) Write(writer io.Writer, value Ttl) { writeInt32(writer, 1) case TtlNoExpiry: writeInt32(writer, 2) - case TtlExpireAfterTicks: + case TtlExpireAfterMillis: writeInt32(writer, 3) FfiConverterUint64INSTANCE.Write(writer, variant_value.Field0) - case TtlExpireAt: + case TtlExpireAtMillis: writeInt32(writer, 4) FfiConverterInt64INSTANCE.Write(writer, variant_value.Field0) default: diff --git a/bindings/go/uniffi/slatedb_test.go b/bindings/go/uniffi/slatedb_test.go index dfcc42dbed..ab07730c4b 100644 --- a/bindings/go/uniffi/slatedb_test.go +++ b/bindings/go/uniffi/slatedb_test.go @@ -3072,7 +3072,7 @@ func TestDbTtl(t *testing.T) { key, value := []byte("alpha"), []byte("one") - putOptions := slatedb.PutOptions{Ttl: slatedb.TtlExpireAt{Field0: 1}} + putOptions := slatedb.PutOptions{Ttl: slatedb.TtlExpireAtMillis{Field0: 1}} writeOptions := slatedb.WriteOptions{AwaitDurable: true} _, err := handle.db.PutWithOptions(key, value, putOptions, writeOptions) if err != nil { @@ -3117,16 +3117,16 @@ type batchSeedRow struct { // needs one extra call returning an empty slice to detect exhaustion. var batchSeedRows = []batchSeedRow{ {key: "batch:01", value: "one", ttl: slatedb.TtlNoExpiry{}}, - {key: "batch:02", value: "two", ttl: slatedb.TtlExpireAfterTicks{Field0: batchSeedTtlTicks}}, + {key: "batch:02", value: "two", ttl: slatedb.TtlExpireAfterMillis{Field0: batchSeedTtlMillis}}, {key: "batch:03", value: "three", ttl: slatedb.TtlNoExpiry{}}, - {key: "batch:04", value: "four", ttl: slatedb.TtlExpireAfterTicks{Field0: batchSeedTtlTicks}}, + {key: "batch:04", value: "four", ttl: slatedb.TtlExpireAfterMillis{Field0: batchSeedTtlMillis}}, {key: "batch:05", value: "five", ttl: slatedb.TtlNoExpiry{}}, - {key: "batch:06", value: "six", ttl: slatedb.TtlExpireAfterTicks{Field0: batchSeedTtlTicks}}, + {key: "batch:06", value: "six", ttl: slatedb.TtlExpireAfterMillis{Field0: batchSeedTtlMillis}}, } -// batchSeedTtlTicks is far enough in the future that TTL'd seed rows never +// batchSeedTtlMillis is far enough in the future that TTL'd seed rows never // expire mid-test, while still producing a non-nil ExpireTs. -const batchSeedTtlTicks = 3_600_000 +const batchSeedTtlMillis = 3_600_000 func seedBatchRows(t *testing.T, db *slatedb.Db) { t.Helper() diff --git a/bindings/uniffi/src/config.rs b/bindings/uniffi/src/config.rs index ecebff4e57..9abed81d5e 100644 --- a/bindings/uniffi/src/config.rs +++ b/bindings/uniffi/src/config.rs @@ -102,10 +102,10 @@ pub enum Ttl { Default, /// Store the value without expiration. NoExpiry, - /// Expire the value after the given number of clock ticks. - ExpireAfterTicks(u64), - /// Expire the value at the given absolute timestamp (clock ticks). - ExpireAt(i64), + /// Expire the value after the given number of milliseconds. + ExpireAfterMillis(u64), + /// Expire the value at the given Unix timestamp in milliseconds. + ExpireAtMillis(i64), } impl From for slatedb::config::Ttl { @@ -113,8 +113,8 @@ impl From for slatedb::config::Ttl { match value { Ttl::Default => Self::Default, Ttl::NoExpiry => Self::NoExpiry, - Ttl::ExpireAfterTicks(ttl) => Self::ExpireAfter(ttl), - Ttl::ExpireAt(ts) => Self::ExpireAt(ts), + Ttl::ExpireAfterMillis(ttl_millis) => Self::ExpireAfterMillis(ttl_millis), + Ttl::ExpireAtMillis(timestamp_millis) => Self::ExpireAtMillis(timestamp_millis), } } } diff --git a/bindings/uniffi/src/settings.rs b/bindings/uniffi/src/settings.rs index c4fa3f871e..d17975f86b 100644 --- a/bindings/uniffi/src/settings.rs +++ b/bindings/uniffi/src/settings.rs @@ -94,8 +94,8 @@ impl Settings { /// Examples: /// /// - `set("flush_interval", "\"250ms\"")` - /// - `set("default_ttl", "42")` - /// - `set("default_ttl", "null")` + /// - `set("default_ttl_millis", "42")` + /// - `set("default_ttl_millis", "null")` /// - `set("compactor_options.max_sst_size", "33554432")` /// - `set("object_store_cache_options.root_folder", "\"/tmp/slatedb-cache\"")` pub fn set(&self, key: String, value_json: String) -> Result<(), Error> { @@ -271,13 +271,13 @@ mod tests { let settings = Arc::new(Settings::new(slatedb::Settings::default())); settings - .set("default_ttl".to_owned(), "100".to_owned()) + .set("default_ttl_millis".to_owned(), "100".to_owned()) .unwrap(); settings - .set("default_ttl".to_owned(), "null".to_owned()) + .set("default_ttl_millis".to_owned(), "null".to_owned()) .unwrap(); - assert_eq!(settings.inner().default_ttl, None); + assert_eq!(settings.inner().default_ttl_millis, None); } #[test] @@ -315,13 +315,13 @@ mod tests { let settings = Arc::new(Settings::new(slatedb::Settings::default())); settings - .set("default_ttl".to_owned(), "42".to_owned()) + .set("default_ttl_millis".to_owned(), "42".to_owned()) .unwrap(); let encoded = settings.to_json_string().unwrap(); let decoded = Settings::from_json_string(encoded).unwrap(); - assert_eq!(decoded.inner().default_ttl, Some(42)); + assert_eq!(decoded.inner().default_ttl_millis, Some(42)); } #[test] @@ -367,7 +367,7 @@ flush_interval = "1s" #[test] fn settings_from_env_with_default_uses_default_snapshot() { figment::Jail::expect_with(|jail| { - jail.set_env("FFI_SETTINGS_DEFAULT_TTL", "42"); + jail.set_env("FFI_SETTINGS_DEFAULT_TTL_MILLIS", "42"); let defaults = Arc::new(Settings::new(slatedb::Settings::default())); defaults @@ -382,7 +382,7 @@ flush_interval = "1s" settings.inner().flush_interval, Some(Duration::from_millis(250)) ); - assert_eq!(settings.inner().default_ttl, Some(42)); + assert_eq!(settings.inner().default_ttl_millis, Some(42)); Ok(()) }); diff --git a/rfcs/0003-timestamps-and-ttl.md b/rfcs/0003-timestamps-and-ttl.md index f63b9bee48..0262279836 100644 --- a/rfcs/0003-timestamps-and-ttl.md +++ b/rfcs/0003-timestamps-and-ttl.md @@ -93,11 +93,11 @@ tick and sleeping briefly if clock skew is detected. pub struct DbOptions { // ... - /// The default time-to-live (TTL) for insertions (note that re-inserting a key - /// with any value will update the TTL to use the default_ttl) + /// The default time-to-live (TTL), in milliseconds, for insertions (note that + /// re-inserting a key with any value will update the TTL to use default_ttl_millis) /// /// Default: no TTL (insertions will remain until deleted) - default_ttl: Option + default_ttl_millis: Option } ``` @@ -126,7 +126,7 @@ pub enum Ttl { /// No expiration for this entry NoExpiry, /// Expire after the specified duration (in milliseconds) - ExpireAfter(u64), + ExpireAfterMillis(u64), } pub struct PutOptions { @@ -367,4 +367,4 @@ the original `seq0` insert as it logically happened "after" `seq1`. for testing or non-standard time sources. - Updated `DbOptions` to remove the `clock` configuration option - Updated `WriteOptions` to `PutOptions` with a `Ttl` enum that supports `Default`, `NoExpiry`, - and `ExpireAfter(u64)` variants + and `ExpireAfterMillis(u64)` variants diff --git a/rfcs/0006-merge-operator.md b/rfcs/0006-merge-operator.md index 1bf4b1f323..d3f873dffc 100644 --- a/rfcs/0006-merge-operator.md +++ b/rfcs/0006-merge-operator.md @@ -220,12 +220,12 @@ impl DbOptions { ### Extending TTL support -The last public API change is extending the `Ttl` enum to support a new `ExpireAt(ts)` variant. This allows users to set a specific expiration time for a key, which overrides the default TTL behavior. +The last public API change is extending the `Ttl` enum to support a new `ExpireAtMillis(timestamp_millis)` variant. This allows users to set a specific expiration time for a key, expressed as milliseconds since the Unix epoch, which overrides the default TTL behavior. ```rust pub enum Ttl { ... - ExpireAt(i64), + ExpireAtMillis(i64), } ``` @@ -425,8 +425,8 @@ SlateDB supports two TTL approaches: 1. **Operation-Level TTL** - Each operation (put/merge) has its own independent TTL, specified via: - - `Ttl::ExpireAfter(duration)`: Expires after specified duration (internally this is implemented as `ExpireAt(Instant::now() + duration)`) - - `Ttl::ExpireAt(timestamp)`: Expires at specified timestamp + - `Ttl::ExpireAfterMillis(duration_millis)`: Expires after the specified duration in milliseconds (internally this is implemented as `create_ts + duration_millis`) + - `Ttl::ExpireAtMillis(timestamp_millis)`: Expires at the specified Unix timestamp in milliseconds - Enables per-element expiration in collections 2. **TTL Renewal (NOT SUPPORTED NATIVELY)** @@ -439,7 +439,7 @@ When merging values with different TTLs, the merge operation only combines value Unlike regular values which become tombstones upon expiration, expired merge entries are simply removed, enabling per-element expiration in collections. -Users can implement custom TTL patterns by consistently using either `ExpireAt` or `ExpireAfter` across operations. +Users can implement custom TTL patterns by consistently using either `ExpireAtMillis` or `ExpireAfterMillis` across operations. ### Ordering Guarantees @@ -498,4 +498,4 @@ One possible optimization is to introduce a new read option for persisting the r It might be useful to have a read option that allows for early termination of merge operations. This could be beneficial for buffering use cases where the user wants to limit the number of operands that are merged together (effectively allowing partial iterations). -RocksDB provides an alternative approach by exposing `GetMergeOperands` which allows listing the unmerged operands directly. \ No newline at end of file +RocksDB provides an alternative approach by exposing `GetMergeOperands` which allows listing the unmerged operands directly. diff --git a/slatedb/benches/db_transaction.rs b/slatedb/benches/db_transaction.rs index 8954209b3b..c0230ebcce 100644 --- a/slatedb/benches/db_transaction.rs +++ b/slatedb/benches/db_transaction.rs @@ -53,13 +53,13 @@ fn concat_merge_operator() -> Arc { fn put_options() -> PutOptions { PutOptions { - ttl: Ttl::ExpireAfter(3_600), + ttl: Ttl::ExpireAfterMillis(3_600_000), } } fn merge_options() -> MergeOptions { MergeOptions { - ttl: Ttl::ExpireAfter(3_600), + ttl: Ttl::ExpireAfterMillis(3_600_000), } } diff --git a/slatedb/benches/write_batch.rs b/slatedb/benches/write_batch.rs index 2bb2c11ea6..9bb206922d 100644 --- a/slatedb/benches/write_batch.rs +++ b/slatedb/benches/write_batch.rs @@ -56,13 +56,13 @@ fn size_sum_merge_operator() -> Arc { fn put_options() -> PutOptions { PutOptions { - ttl: Ttl::ExpireAfter(3_600), + ttl: Ttl::ExpireAfterMillis(3_600_000), } } fn merge_options() -> MergeOptions { MergeOptions { - ttl: Ttl::ExpireAfter(3_600), + ttl: Ttl::ExpireAfterMillis(3_600_000), } } diff --git a/slatedb/src/batch.rs b/slatedb/src/batch.rs index 3b2e03dcd1..43b75262d2 100644 --- a/slatedb/src/batch.rs +++ b/slatedb/src/batch.rs @@ -309,7 +309,7 @@ impl WriteBatch { &self, seq: u64, now: i64, - default_ttl: Option, + default_ttl_millis: Option, merger: Option, extractor: Option<&dyn PrefixExtractor>, ) -> Result<(Vec, BTreeSet, u64), SlateDBError> { @@ -320,7 +320,7 @@ impl WriteBatch { IterationOrder::Ascending, seq, Some(now), - default_ttl, + default_ttl_millis, )); if self.has_merge_ops() { if let Some(ref merge_operator) = merger { @@ -379,12 +379,12 @@ pub mod benches { batch: &WriteBatch, seq: u64, now: i64, - default_ttl: Option, + default_ttl_millis: Option, merger: Option>, extractor: Option<&dyn PrefixExtractor>, ) -> Result<(Vec, BTreeSet, u64), Error> { batch - .extract_entries(seq, now, default_ttl, merger, extractor) + .extract_entries(seq, now, default_ttl_millis, merger, extractor) .await .map_err(Into::into) } @@ -410,11 +410,11 @@ impl WriteBatchIterator { op: &WriteOp, seq: u64, now: Option, - default_ttl: Option, + default_ttl_millis: Option, ) -> RowEntry { let expire_ts = match (op, now) { - (WriteOp::Put(_, opts), Some(now)) => opts.expire_ts_from(default_ttl, now), - (WriteOp::Merge(_, opts), Some(now)) => opts.expire_ts_from(default_ttl, now), + (WriteOp::Put(_, opts), Some(now)) => opts.expire_ts_from(default_ttl_millis, now), + (WriteOp::Merge(_, opts), Some(now)) => opts.expire_ts_from(default_ttl_millis, now), _ => None, }; op.to_row_entry(key, seq, now, expire_ts) @@ -426,21 +426,21 @@ impl WriteBatchIterator { ordering: IterationOrder, seq: u64, now: Option, - default_ttl: Option, + default_ttl_millis: Option, ) -> Self { let entries: Vec = match ordering { IterationOrder::Ascending => batch .ops .range(range) .flat_map(|(key, ops)| ops.iter().rev().map(move |op| (key, op))) - .map(|(key, op)| Self::write_op_to_row_entry(key, op, seq, now, default_ttl)) + .map(|(key, op)| Self::write_op_to_row_entry(key, op, seq, now, default_ttl_millis)) .collect(), IterationOrder::Descending => batch .ops .range(range) .rev() .flat_map(|(key, ops)| ops.iter().rev().map(move |op| (key, op))) - .map(|(key, op)| Self::write_op_to_row_entry(key, op, seq, now, default_ttl)) + .map(|(key, op)| Self::write_op_to_row_entry(key, op, seq, now, default_ttl_millis)) .collect(), }; @@ -853,7 +853,7 @@ mod tests { // Given: an empty WriteBatch and custom merge options let mut batch = WriteBatch::new(); let merge_options = MergeOptions { - ttl: Ttl::ExpireAfter(3600), // 1 hour + ttl: Ttl::ExpireAfterMillis(3_600_000), // 1 hour }; // When: adding a merge operation with custom options @@ -865,7 +865,7 @@ mod tests { match op { WriteOp::Merge(value, options) => { assert_eq!(value.as_ref(), b"value1"); - assert_eq!(options.ttl, Ttl::ExpireAfter(3600)); + assert_eq!(options.ttl, Ttl::ExpireAfterMillis(3_600_000)); } _ => panic!("Expected Merge operation"), } @@ -1422,14 +1422,14 @@ mod tests { b"key1", b"a", &MergeOptions { - ttl: Ttl::ExpireAfter(3600), + ttl: Ttl::ExpireAfterMillis(3600), }, ); batch.merge_with_options( b"key1", b"b", &MergeOptions { - ttl: Ttl::ExpireAt(4600), + ttl: Ttl::ExpireAtMillis(4600), }, ); @@ -1456,14 +1456,14 @@ mod tests { b"key1", b"a", &MergeOptions { - ttl: Ttl::ExpireAfter(3600), + ttl: Ttl::ExpireAfterMillis(3600), }, ); batch.merge_with_options( b"key1", b"b", &MergeOptions { - ttl: Ttl::ExpireAfter(7200), + ttl: Ttl::ExpireAfterMillis(7200), }, ); @@ -1495,14 +1495,14 @@ mod tests { b"key1", b"a", &MergeOptions { - ttl: Ttl::ExpireAfter(3600), + ttl: Ttl::ExpireAfterMillis(3600), }, ); batch.merge_with_options( b"key2", b"b", &MergeOptions { - ttl: Ttl::ExpireAfter(7200), + ttl: Ttl::ExpireAfterMillis(7200), }, ); @@ -1528,7 +1528,7 @@ mod tests { b"key1", b"a", &MergeOptions { - ttl: Ttl::ExpireAfter(7200), + ttl: Ttl::ExpireAfterMillis(7200), }, ); @@ -1550,7 +1550,7 @@ mod tests { b"key1", b"a", &MergeOptions { - ttl: Ttl::ExpireAfter(7200), + ttl: Ttl::ExpireAfterMillis(7200), }, ); diff --git a/slatedb/src/batch_write.rs b/slatedb/src/batch_write.rs index 1749177d46..00a067fd13 100644 --- a/slatedb/src/batch_write.rs +++ b/slatedb/src/batch_write.rs @@ -246,7 +246,7 @@ impl DbInner { .extract_entries( commit_seq, now, - self.settings.default_ttl, + self.settings.default_ttl_millis, self.flush_merge_operator.clone(), self.segment_extractor.as_deref(), ) diff --git a/slatedb/src/compactor.rs b/slatedb/src/compactor.rs index 7fb86a66e6..0611510f78 100644 --- a/slatedb/src/compactor.rs +++ b/slatedb/src/compactor.rs @@ -3151,7 +3151,7 @@ mod tests { b"key1", &[b'a'; 32], &crate::config::MergeOptions { - ttl: Ttl::ExpireAfter(10), + ttl: Ttl::ExpireAfterMillis(10), }, &WriteOptions { await_durable: true, @@ -3343,7 +3343,7 @@ mod tests { b"key1", b"a", &crate::config::MergeOptions { - ttl: Ttl::ExpireAfter(100), + ttl: Ttl::ExpireAfterMillis(100), }, &WriteOptions { await_durable: false, @@ -3369,7 +3369,7 @@ mod tests { b"key1", b"b", &crate::config::MergeOptions { - ttl: Ttl::ExpireAfter(200), + ttl: Ttl::ExpireAfterMillis(200), }, &WriteOptions { await_durable: false, @@ -3478,13 +3478,13 @@ mod tests { flush_type: FlushType::MemTable, }; - // write merge operations with the SAME ExpireAt timestamp at different clock times + // write merge operations with the SAME ExpireAtMillis timestamp at different clock times system_clock.set(100); db.merge_with_options( b"key1", b"a", &MergeOptions { - ttl: Ttl::ExpireAt(1000), + ttl: Ttl::ExpireAtMillis(1000), }, &WriteOptions { await_durable: false, @@ -3500,7 +3500,7 @@ mod tests { b"key1", b"b", &MergeOptions { - ttl: Ttl::ExpireAt(1000), + ttl: Ttl::ExpireAtMillis(1000), }, &WriteOptions { await_durable: false, @@ -3601,7 +3601,7 @@ mod tests { &[1; 16], value, &PutOptions { - ttl: Ttl::ExpireAt(10), + ttl: Ttl::ExpireAtMillis(10), }, &WriteOptions { await_durable: false, @@ -3617,7 +3617,7 @@ mod tests { &[2; 16], value, &PutOptions { - ttl: Ttl::ExpireAt(i64::MAX), + ttl: Ttl::ExpireAtMillis(i64::MAX), }, &WriteOptions { await_durable: false, @@ -3694,7 +3694,7 @@ mod tests { } .into(); let mut options = db_options(Some(compactor_options())); - options.default_ttl = Some(50); + options.default_ttl_millis = Some(50); options .compactor_options .as_mut() @@ -3717,7 +3717,7 @@ mod tests { &[1; 16], value, &PutOptions { - ttl: Ttl::ExpireAfter(10), + ttl: Ttl::ExpireAfterMillis(10), }, &WriteOptions { await_durable: false, @@ -3764,7 +3764,7 @@ mod tests { &[1; 16], value, &PutOptions { - ttl: Ttl::ExpireAfter(80), + ttl: Ttl::ExpireAfterMillis(80), }, &WriteOptions { await_durable: false, diff --git a/slatedb/src/config.rs b/slatedb/src/config.rs index a1dcc46ced..483245c839 100644 --- a/slatedb/src/config.rs +++ b/slatedb/src/config.rs @@ -503,25 +503,29 @@ pub struct PutOptions { } impl PutOptions { - pub(crate) fn expire_ts_from(&self, default: Option, now: i64) -> Option { + pub(crate) fn expire_ts_from( + &self, + default_ttl_millis: Option, + now_millis: i64, + ) -> Option { match self.ttl { - Ttl::Default => match default { + Ttl::Default => match default_ttl_millis { None => None, - Some(default_ttl) => Self::checked_expire_ts(now, default_ttl), + Some(default_ttl_millis) => Self::checked_expire_ts(now_millis, default_ttl_millis), }, Ttl::NoExpiry => None, - Ttl::ExpireAfter(ttl) => Self::checked_expire_ts(now, ttl), - Ttl::ExpireAt(ts) => Some(ts), + Ttl::ExpireAfterMillis(ttl_millis) => Self::checked_expire_ts(now_millis, ttl_millis), + Ttl::ExpireAtMillis(timestamp_millis) => Some(timestamp_millis), } } - fn checked_expire_ts(now: i64, ttl: u64) -> Option { + fn checked_expire_ts(now_millis: i64, ttl_millis: u64) -> Option { // for overflow, we will just assume no TTL - if ttl > i64::MAX as u64 { + if ttl_millis > i64::MAX as u64 { return None; }; - let expire_ts = now + (ttl as i64); - if expire_ts < now { + let expire_ts = now_millis + (ttl_millis as i64); + if expire_ts < now_millis { return None; }; @@ -541,25 +545,29 @@ pub struct MergeOptions { impl MergeOptions { // TODO(agavra): deduplicate this with PutOptions::expire_ts_from - pub(crate) fn expire_ts_from(&self, default: Option, now: i64) -> Option { + pub(crate) fn expire_ts_from( + &self, + default_ttl_millis: Option, + now_millis: i64, + ) -> Option { match self.ttl { - Ttl::Default => match default { + Ttl::Default => match default_ttl_millis { None => None, - Some(default_ttl) => Self::checked_expire_ts(now, default_ttl), + Some(default_ttl_millis) => Self::checked_expire_ts(now_millis, default_ttl_millis), }, Ttl::NoExpiry => None, - Ttl::ExpireAfter(ttl) => Self::checked_expire_ts(now, ttl), - Ttl::ExpireAt(ts) => Some(ts), + Ttl::ExpireAfterMillis(ttl_millis) => Self::checked_expire_ts(now_millis, ttl_millis), + Ttl::ExpireAtMillis(timestamp_millis) => Some(timestamp_millis), } } - fn checked_expire_ts(now: i64, ttl: u64) -> Option { + fn checked_expire_ts(now_millis: i64, ttl_millis: u64) -> Option { // for overflow, we will just assume no TTL - if ttl > i64::MAX as u64 { + if ttl_millis > i64::MAX as u64 { return None; }; - let expire_ts = now + (ttl as i64); - if expire_ts < now { + let expire_ts = now_millis + (ttl_millis as i64); + if expire_ts < now_millis { return None; }; @@ -567,14 +575,25 @@ impl MergeOptions { } } +/// Time-to-live policy applied to an inserted value or merge operand. +/// +/// TTL durations are expressed in milliseconds. Absolute expiration timestamps are +/// expressed as milliseconds since the Unix epoch. +/// +/// Expiration is applied during compaction and is therefore best effort; an expired +/// value may remain visible until compaction processes it. #[non_exhaustive] #[derive(Clone, Default, PartialEq, Debug)] pub enum Ttl { + /// Use [`Settings::default_ttl_millis`]. #[default] Default, + /// Store the value without an expiration. NoExpiry, - ExpireAfter(u64), - ExpireAt(i64), + /// Expire the value after the specified number of milliseconds. + ExpireAfterMillis(u64), + /// Expire the value at the specified Unix timestamp in milliseconds. + ExpireAtMillis(i64), } /// Defines the scope targeted by a given checkpoint. If set to All, then the checkpoint will @@ -765,11 +784,12 @@ pub struct Settings { #[serde(default)] pub metric_level: MetricLevel, - /// The default time-to-live (TTL) for insertions (note that re-inserting a key - /// with any value will update the TTL to use the default_ttl) + /// The default time-to-live (TTL), in milliseconds, for insertions (note that + /// re-inserting a key with any value will update the TTL to use + /// `default_ttl_millis`). /// /// Default: no TTL (insertions will remain until deleted) - pub default_ttl: Option, + pub default_ttl_millis: Option, /// Maximum number of wrapper-level retries for a single object-store /// operation, on top of the `object_store` client's own HTTP retries. @@ -819,7 +839,7 @@ impl std::fmt::Debug for Settings { ) .field("garbage_collector_options", &self.garbage_collector_options) .field("metric_level", &self.metric_level) - .field("default_ttl", &self.default_ttl); + .field("default_ttl_millis", &self.default_ttl_millis); data.finish() } } @@ -1049,7 +1069,7 @@ impl Default for Settings { object_store_cache_options: ObjectStoreCacheOptions::default(), garbage_collector_options: Some(GarbageCollectorOptions::default()), metric_level: MetricLevel::default(), - default_ttl: None, + default_ttl_millis: None, object_store_max_retries: None, #[cfg(test)] block_format: None, @@ -1991,7 +2011,7 @@ object_store_cache_options: fn should_return_exact_timestamp_for_put_expire_at() { // given let opts = PutOptions { - ttl: Ttl::ExpireAt(12345), + ttl: Ttl::ExpireAtMillis(12345), }; // when @@ -2005,7 +2025,7 @@ object_store_cache_options: fn should_ignore_default_ttl_for_put_expire_at() { // given let opts = PutOptions { - ttl: Ttl::ExpireAt(12345), + ttl: Ttl::ExpireAtMillis(12345), }; // when @@ -2019,7 +2039,7 @@ object_store_cache_options: fn should_allow_past_timestamp_for_put_expire_at() { // given let opts = PutOptions { - ttl: Ttl::ExpireAt(50), + ttl: Ttl::ExpireAtMillis(50), }; // when @@ -2033,7 +2053,7 @@ object_store_cache_options: fn should_return_exact_timestamp_for_merge_expire_at() { // given let opts = MergeOptions { - ttl: Ttl::ExpireAt(12345), + ttl: Ttl::ExpireAtMillis(12345), }; // when @@ -2045,9 +2065,9 @@ object_store_cache_options: #[test] fn should_return_deterministic_expire_ts_for_expire_at() { - // given: same ExpireAt value used at different times + // given: same ExpireAtMillis value used at different times let opts = PutOptions { - ttl: Ttl::ExpireAt(99999), + ttl: Ttl::ExpireAtMillis(99999), }; // when diff --git a/slatedb/src/db.rs b/slatedb/src/db.rs index df23172470..d6782ef363 100644 --- a/slatedb/src/db.rs +++ b/slatedb/src/db.rs @@ -7175,7 +7175,7 @@ mod tests { min_filter_keys: u32, l0_sst_size_bytes: usize, compactor_options: Option, - ttl: Option, + default_ttl_millis: Option, ) -> Settings { Settings { flush_interval: Some(Duration::from_millis(100)), @@ -7195,7 +7195,7 @@ mod tests { object_store_cache_options: ObjectStoreCacheOptions::default(), garbage_collector_options: None, metric_level: MetricLevel::default(), - default_ttl: ttl, + default_ttl_millis, object_store_max_retries: None, block_format: None, } @@ -8126,7 +8126,7 @@ mod tests { // Put with options (TTL) clock.set(200); let put_opts = PutOptions { - ttl: Ttl::ExpireAfter(1000), + ttl: Ttl::ExpireAfterMillis(1000), }; let handle = db .put_with_options( @@ -9408,7 +9408,7 @@ mod tests { key, value, &PutOptions { - ttl: Ttl::ExpireAfter(50), + ttl: Ttl::ExpireAfterMillis(50), }, &WriteOptions { await_durable: false, @@ -9441,7 +9441,7 @@ mod tests { .unwrap(); let put_opts = PutOptions { - ttl: Ttl::ExpireAfter(50), + ttl: Ttl::ExpireAfterMillis(50), }; let write_opts = WriteOptions { await_durable: false, @@ -9511,13 +9511,13 @@ mod tests { .await .unwrap(); - // when: write with ExpireAt at different clock times + // when: write with ExpireAtMillis at different clock times clock.set(100); db.put_with_options( b"key1", b"value1", &PutOptions { - ttl: Ttl::ExpireAt(500), + ttl: Ttl::ExpireAtMillis(500), }, &WriteOptions { await_durable: false, @@ -9532,7 +9532,7 @@ mod tests { b"key2", b"value2", &PutOptions { - ttl: Ttl::ExpireAt(500), + ttl: Ttl::ExpireAtMillis(500), }, &WriteOptions { await_durable: false, @@ -9701,14 +9701,14 @@ mod tests { b"key1", b"a", &MergeOptions { - ttl: Ttl::ExpireAfter(3600), + ttl: Ttl::ExpireAfterMillis(3600), }, ); batch.merge_with_options( b"key1", b"b", &MergeOptions { - ttl: Ttl::ExpireAfter(7200), + ttl: Ttl::ExpireAfterMillis(7200), }, ); diff --git a/slatedb/src/db_transaction.rs b/slatedb/src/db_transaction.rs index ff5d494927..0c25f2bb3f 100644 --- a/slatedb/src/db_transaction.rs +++ b/slatedb/src/db_transaction.rs @@ -2050,7 +2050,7 @@ mod tests { b"counter", 1u64.to_le_bytes(), &MergeOptions { - ttl: crate::config::Ttl::ExpireAfter(3600), + ttl: crate::config::Ttl::ExpireAfterMillis(3600), }, ) .unwrap(); @@ -2058,7 +2058,7 @@ mod tests { b"counter", 2u64.to_le_bytes(), &MergeOptions { - ttl: crate::config::Ttl::ExpireAfter(7200), + ttl: crate::config::Ttl::ExpireAfterMillis(7200), }, ) .unwrap(); @@ -2098,7 +2098,7 @@ mod tests { object_store_cache_options: crate::config::ObjectStoreCacheOptions::default(), garbage_collector_options: None, metric_level: MetricLevel::default(), - default_ttl: None, + default_ttl_millis: None, object_store_max_retries: None, block_format: None, } @@ -2137,7 +2137,7 @@ mod tests { clock.set(200); let txn = db.begin(IsolationLevel::Snapshot).await.unwrap(); let put_opts = PutOptions { - ttl: crate::config::Ttl::ExpireAfter(1000), + ttl: crate::config::Ttl::ExpireAfterMillis(1000), }; txn.put_with_options(b"key2", b"value2", &put_opts).unwrap(); let handle = txn diff --git a/website/src/content/docs/docs/design/time.mdx b/website/src/content/docs/docs/design/time.mdx index 29941251b5..28704b4369 100644 --- a/website/src/content/docs/docs/design/time.mdx +++ b/website/src/content/docs/docs/design/time.mdx @@ -33,18 +33,18 @@ SlateDB also preserves the same metadata in the WAL path, which is why [Change D ## Expiration -TTL is stored as an absolute expiration timestamp, not a relative duration. On commit, SlateDB computes `expire_ts = create_ts + ttl`. +TTL is stored as an absolute expiration timestamp in milliseconds since the Unix epoch, not as a relative duration. On commit, SlateDB computes `expire_ts = create_ts + ttl_millis`. -[`Settings::default_ttl`](https://docs.rs/slatedb/latest/slatedb/config/struct.Settings.html#structfield.default_ttl) sets a default TTL for puts and merges. [`PutOptions`](https://docs.rs/slatedb/latest/slatedb/config/struct.PutOptions.html) and [`MergeOptions`](https://docs.rs/slatedb/latest/slatedb/config/struct.MergeOptions.html) can override that per operation: +[`Settings::default_ttl_millis`](https://docs.rs/slatedb/latest/slatedb/config/struct.Settings.html#structfield.default_ttl_millis) sets a default TTL in milliseconds for puts and merges. [`PutOptions`](https://docs.rs/slatedb/latest/slatedb/config/struct.PutOptions.html) and [`MergeOptions`](https://docs.rs/slatedb/latest/slatedb/config/struct.MergeOptions.html) can override that per operation: - [`Ttl::NoExpiry`](https://docs.rs/slatedb/latest/slatedb/config/enum.Ttl.html#variant.NoExpiry) — store the value without expiration -- [`Ttl::ExpireAfter(u64)`](https://docs.rs/slatedb/latest/slatedb/config/enum.Ttl.html#variant.ExpireAfter) — expire after a relative duration (clock ticks) -- [`Ttl::ExpireAt(i64)`](https://docs.rs/slatedb/latest/slatedb/config/enum.Ttl.html#variant.ExpireAt) — expire at a fixed absolute timestamp (clock ticks) +- [`Ttl::ExpireAfterMillis(u64)`](https://docs.rs/slatedb/latest/slatedb/config/enum.Ttl.html#variant.ExpireAfterMillis) — expire after a relative duration in milliseconds +- [`Ttl::ExpireAtMillis(i64)`](https://docs.rs/slatedb/latest/slatedb/config/enum.Ttl.html#variant.ExpireAtMillis) — expire at a fixed Unix timestamp in milliseconds Deletes write tombstones and do not carry TTL. :::caution -Batch-local merges requires one effective `expire_ts` per key. `Db::write(...)` and transaction commit return `ErrorKind::Invalid` if a single batch tries to merge the same key across different expiration timestamps. Effective timestamps are computed after applying `Settings::default_ttl`, so `Ttl::Default` can still differ from `Ttl::NoExpiry` or from another explicit TTL. +Batch-local merges requires one effective `expire_ts` per key. `Db::write(...)` and transaction commit return `ErrorKind::Invalid` if a single batch tries to merge the same key across different expiration timestamps. Effective timestamps are computed after applying `Settings::default_ttl_millis`, so `Ttl::Default` can still differ from `Ttl::NoExpiry` or from another explicit TTL. ::: :::note From 38df98a93566a4acb5308eec225f8b697f567da3 Mon Sep 17 00:00:00 2001 From: Rohan Date: Thu, 30 Jul 2026 01:52:04 -0400 Subject: [PATCH 05/65] remove notifiers from WalBuffer (#1987) --- slatedb/src/wal_buffer.rs | 110 ++++---------------------------------- 1 file changed, 11 insertions(+), 99 deletions(-) diff --git a/slatedb/src/wal_buffer.rs b/slatedb/src/wal_buffer.rs index 543eb79b3a..72339058ca 100644 --- a/slatedb/src/wal_buffer.rs +++ b/slatedb/src/wal_buffer.rs @@ -10,7 +10,7 @@ use crate::error::SlateDBError; use crate::tablestore::TableStore; use crate::types::RowEntry; use crate::utils::SafeSender; -use crate::utils::{format_bytes_si, WatchableOnceCell, WatchableOnceCellReader}; +use crate::utils::{format_bytes_si, WatchableOnceCellReader}; use crate::wal; use crate::wal::{FlushResultFuture, WalError, WalEvent, WalStatus, WalWriter}; use crate::wal_buffer_stats::WalBufferStats; @@ -92,8 +92,6 @@ struct WalBufferManagerInner { struct WalBuffer { /// queue for the entries entries: VecDeque, - /// watcher to await durability - durable: WatchableOnceCell>, /// the sequence number of the most recent addition to this WAL buffer last_seq: u64, /// size of the entries that has been added to the WAL buffer in bytes @@ -154,20 +152,17 @@ impl WalBufferManager { }) } - //TODO: do we still need durable watchers here? /// Check if we need to flush the wal with considering max_wal_size. the checking over `max_wal_size` /// is not very strict, we have to ensure a write batch into a single WAL file. /// /// It's the caller's duty to call `maybe_trigger_flush` after calling `append`. - fn maybe_trigger_flush( - &self, - ) -> Result>, WalError> { - let (durable_watcher, need_flush, flush_epoch) = { + fn maybe_trigger_flush(&self) -> Result<(), WalError> { + let (need_flush, flush_epoch) = { let inner = self.inner.read(); // checks the size of the current wal let (need_flush, flush_epoch) = inner.needs_flush(&self.table_store, self.max_wal_bytes_size); - (inner.current_wal.durable_watcher(), need_flush, flush_epoch) + (need_flush, flush_epoch) }; if need_flush { // Only send a flush request if one hasn't already been sent for this epoch. @@ -187,7 +182,7 @@ impl WalBufferManager { self.stats .estimated_bytes .set(status.estimated_bytes as i64); - Ok(durable_watcher) + Ok(()) } /// Send a flush request to the background flush worker. @@ -243,7 +238,7 @@ impl WalWriter for WalBufferManager { }; self.inner .write() - .drain_on_close(WalError::Closed, &self.table_store); + .mark_closed(WalError::Closed, &self.table_store); Ok(()) } } @@ -334,17 +329,11 @@ impl WalBufferManagerInner { } } - fn drain_on_close( - &mut self, - reason: WalError, - table_store: &TableStore, - ) -> (WalStatus, Vec<(u64, Arc)>) { + fn mark_closed(&mut self, reason: WalError, table_store: &TableStore) -> WalStatus { self.flush_task_exited_reason = Some(reason); self.freeze_current_wal(); - let unflushed_wals = self.flushing_wals(); self.immutable_wals.clear(); - let status = self.compute_status(table_store); - (status, unflushed_wals) + self.compute_status(table_store) } fn freeze_current_wal(&mut self) { @@ -388,7 +377,6 @@ impl WalBuffer { fn new() -> Self { Self { entries: VecDeque::new(), - durable: WatchableOnceCell::new(), last_seq: 0, entries_size: 0, } @@ -405,22 +393,6 @@ impl WalBuffer { WalBufferIterator::new(self) } - /// Returns a watcher that can be used to await durability. - fn durable_watcher(&self) -> WatchableOnceCellReader> { - self.durable.reader() - } - - /// Awaits until the WAL is durable (flushed to storage). - #[cfg(test)] - async fn await_durable(&self) -> Result<(), SlateDBError> { - self.durable.reader().await_value().await - } - - /// Notifies that the WAL has been made durable (or failed). - fn notify_durable(&self, result: Result<(), SlateDBError>) { - self.durable.write(result); - } - /// Returns true if the buffer is empty. fn is_empty(&self) -> bool { self.entries.is_empty() @@ -513,18 +485,11 @@ impl WalFlushHandler { inner.compute_status(&self.table_store) }; - // we notify the listener first since that updates the oracle, and then notify - // the table waiters. blocked writes wait on the table, so we have to update the oracle - // first to preserve read-your-writes. This does mean that there is a small window - // after notifying flushed before the wal memory is actually released. - // TODO: once we change writes to block on the durable seq num from the oracle we - // can simplify this and fully drop the wal before notifying listeners - self.notify_listener(wal::WalEvent::WalFlushed(status)); - wal.notify_durable(result.clone()); if Arc::strong_count(&wal) > 1 { warn!("outstanding references to wal id {} after flushing", wal_id); } drop(wal); + self.notify_listener(wal::WalEvent::WalFlushed(status)); } Ok(()) @@ -598,10 +563,10 @@ impl MessageHandler for WalFlushHandler { .map(WalError::from) .unwrap_or(WalError::Closed); - let (final_status, unflushed) = self + let final_status = self .inner .write() - .drain_on_close(error.clone(), &self.table_store); + .mark_closed(error.clone(), &self.table_store); self.notify_listener(WalEvent::WalClosed(final_status.clone())); // drain remaining messages @@ -617,13 +582,6 @@ impl MessageHandler for WalFlushHandler { } } } - - // notify all the flushing wals to be finished with fatal error or shutdown - // error. we need ensure all the wal tables finally get notified. freeze current - // WAL to notify writers in the subsequent flushing_wals loop. - for (_, wal) in unflushed { - wal.notify_durable(Err(result.clone().err().unwrap_or(SlateDBError::Closed))); - } Ok(()) } } @@ -762,52 +720,6 @@ mod tests { assert_eq!(buffer.last_seq(), Some(40)); } - #[tokio::test] - async fn test_notify_durable_success() { - let mut buffer = WalBuffer::new(); - buffer.append(make_entry("key", "value", 1, None)); - - buffer.notify_durable(Ok(())); - - let result = buffer.await_durable().await; - assert!(result.is_ok()); - } - - #[tokio::test] - async fn test_notify_durable_error() { - let mut buffer = WalBuffer::new(); - buffer.append(make_entry("key", "value", 1, None)); - - buffer.notify_durable(Err(SlateDBError::Closed)); - - let result = buffer.await_durable().await; - assert!(matches!(result, Err(SlateDBError::Closed))); - } - - #[tokio::test] - async fn test_durable_watcher_returns_reader() { - let mut buffer = WalBuffer::new(); - buffer.append(make_entry("key", "value", 1, None)); - - let mut reader = buffer.durable_watcher(); - buffer.notify_durable(Ok(())); - - let result = reader.await_value().await; - assert!(result.is_ok()); - } - - #[tokio::test] - async fn test_notify_durable_only_sets_once() { - let mut buffer = WalBuffer::new(); - buffer.append(make_entry("key", "value", 1, None)); - - buffer.notify_durable(Ok(())); - buffer.notify_durable(Err(SlateDBError::Closed)); - - let result = buffer.await_durable().await; - assert!(result.is_ok()); - } - #[test] fn test_iter() { let mut buffer = WalBuffer::new(); From 8f673c0f97cfa6a4aabf819f50b5d49ebf03c432 Mon Sep 17 00:00:00 2001 From: Chris Date: Thu, 30 Jul 2026 11:11:24 -0700 Subject: [PATCH 06/65] Remove await_durable from WriteOptions (#1985) --- README.md | 2 +- bindings/go/uniffi/doc.go | 4 +- bindings/go/uniffi/slatedb.go | 466 +++++++++++------- bindings/go/uniffi/slatedb.h | 43 ++ bindings/go/uniffi/slatedb_test.go | 101 ++-- .../java/io/slatedb/uniffi/SlateDbDbTest.java | 34 +- bindings/node/tests/admin.test.mjs | 24 +- bindings/node/tests/db.test.mjs | 50 +- bindings/node/tests/support.mjs | 4 +- bindings/python/tests/conftest.py | 2 +- bindings/python/tests/test_admin.py | 8 +- bindings/python/tests/test_db.py | 32 +- bindings/uniffi/src/config.rs | 20 +- bindings/uniffi/src/db.rs | 59 ++- bindings/uniffi/src/db_transaction.rs | 14 +- bindings/uniffi/src/lib.rs | 4 +- bindings/uniffi/src/types.rs | 18 - bindings/uniffi/src/write_handle.rs | 34 ++ examples/src/azure_blob_storage.rs | 1 - examples/src/google_cloud_storage.rs | 1 - slatedb-bencher/src/db.rs | 20 +- slatedb-bencher/src/main.rs | 12 +- slatedb-bencher/src/transactions.rs | 27 +- slatedb-dst/src/actors/bank/mod.rs | 1 - slatedb-dst/src/actors/bank/transfer.rs | 1 - slatedb-dst/src/actors/fencer.rs | 6 +- slatedb-dst/src/actors/workload.rs | 1 - slatedb/benches/db_operations.rs | 1 - slatedb/benches/db_transaction.rs | 1 - slatedb/benches/scan_prefix_bench.rs | 5 +- slatedb/src/admin.rs | 1 - slatedb/src/batch_write.rs | 76 +-- slatedb/src/clone.rs | 9 +- slatedb/src/compactor.rs | 72 +-- slatedb/src/config.rs | 17 +- slatedb/src/db.rs | 382 ++++++++------ slatedb/src/db_cache_manager.rs | 1 - slatedb/src/db_reader.rs | 20 +- slatedb/src/db_snapshot.rs | 1 - slatedb/src/db_status.rs | 67 ++- slatedb/src/db_transaction.rs | 45 +- slatedb/src/fence.rs | 10 +- slatedb/src/ops.rs | 42 +- slatedb/src/test_utils.rs | 20 +- slatedb/src/wal/mod.rs | 2 +- slatedb/tests/db.rs | 3 - slatedb/tests/prefix_filter.rs | 20 +- .../src/content/docs/docs/design/writes.mdx | 8 +- .../src/content/docs/docs/get-started/faq.mdx | 2 +- 49 files changed, 1049 insertions(+), 745 deletions(-) create mode 100644 bindings/uniffi/src/write_handle.rs diff --git a/README.md b/README.md index a15dc595d2..86fb25f8f1 100644 --- a/README.md +++ b/README.md @@ -16,7 +16,7 @@ To mitigate high write API costs (PUTs), SlateDB batches writes. Rather than writing every `put()` call to object storage, MemTables are flushed periodically to object storage as a string-sorted table (SST). The flush interval is configurable. -`put()` returns a `Future` that resolves when the data is durably persisted. Clients that prefer lower latency at the cost of durability can instead use `put_with_options` with `await_durable` set to `false`. +Write operations return a `WriteHandle` after updating the in-memory WAL and MemTable. Call `handle.await_durable().await` to wait for one write to become durable, or call `db.flush().await` to flush all pending writes. To mitigate read latency and read API costs (GETs), SlateDB will use standard LSM-tree caching techniques: in-memory block caches, compression, bloom filters, and local SST disk caches. diff --git a/bindings/go/uniffi/doc.go b/bindings/go/uniffi/doc.go index 4a12200279..a4672ff4de 100644 --- a/bindings/go/uniffi/doc.go +++ b/bindings/go/uniffi/doc.go @@ -100,8 +100,8 @@ // [Db.Delete], and [Db.Merge], plus batch and durability controls through // [PutOptions], [MergeOptions], [WriteOptions], and [FlushOptions]. // -// [WriteHandle] reports metadata assigned to a successful write, including the -// sequence number and creation timestamp. +// [WriteHandle] reports metadata assigned to a successful write and exposes +// [WriteHandle.AwaitDurable] for waiting until that specific write is durable. // // [WriteBatch] collects multiple mutations and applies them atomically through // [Db.Write] or [Db.WriteWithOptions]. Batches are single-use once submitted. diff --git a/bindings/go/uniffi/slatedb.go b/bindings/go/uniffi/slatedb.go index 31daffe0be..6f1a944631 100644 --- a/bindings/go/uniffi/slatedb.go +++ b/bindings/go/uniffi/slatedb.go @@ -799,7 +799,7 @@ func uniffiCheckChecksums() { checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { return C.uniffi_slatedb_uniffi_checksum_method_db_delete() }) - if checksum != 4063 { + if checksum != 29763 { // If this happens try cleaning and rebuilding your project panic("slatedb: uniffi_slatedb_uniffi_checksum_method_db_delete: UniFFI API checksum mismatch") } @@ -808,7 +808,7 @@ func uniffiCheckChecksums() { checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { return C.uniffi_slatedb_uniffi_checksum_method_db_delete_with_options() }) - if checksum != 44744 { + if checksum != 47162 { // If this happens try cleaning and rebuilding your project panic("slatedb: uniffi_slatedb_uniffi_checksum_method_db_delete_with_options: UniFFI API checksum mismatch") } @@ -880,7 +880,7 @@ func uniffiCheckChecksums() { checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { return C.uniffi_slatedb_uniffi_checksum_method_db_merge() }) - if checksum != 28366 { + if checksum != 37097 { // If this happens try cleaning and rebuilding your project panic("slatedb: uniffi_slatedb_uniffi_checksum_method_db_merge: UniFFI API checksum mismatch") } @@ -889,7 +889,7 @@ func uniffiCheckChecksums() { checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { return C.uniffi_slatedb_uniffi_checksum_method_db_merge_with_options() }) - if checksum != 15865 { + if checksum != 37495 { // If this happens try cleaning and rebuilding your project panic("slatedb: uniffi_slatedb_uniffi_checksum_method_db_merge_with_options: UniFFI API checksum mismatch") } @@ -898,7 +898,7 @@ func uniffiCheckChecksums() { checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { return C.uniffi_slatedb_uniffi_checksum_method_db_put() }) - if checksum != 53275 { + if checksum != 2894 { // If this happens try cleaning and rebuilding your project panic("slatedb: uniffi_slatedb_uniffi_checksum_method_db_put: UniFFI API checksum mismatch") } @@ -907,7 +907,7 @@ func uniffiCheckChecksums() { checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { return C.uniffi_slatedb_uniffi_checksum_method_db_put_with_options() }) - if checksum != 37591 { + if checksum != 12036 { // If this happens try cleaning and rebuilding your project panic("slatedb: uniffi_slatedb_uniffi_checksum_method_db_put_with_options: UniFFI API checksum mismatch") } @@ -988,7 +988,7 @@ func uniffiCheckChecksums() { checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { return C.uniffi_slatedb_uniffi_checksum_method_db_write() }) - if checksum != 29016 { + if checksum != 63711 { // If this happens try cleaning and rebuilding your project panic("slatedb: uniffi_slatedb_uniffi_checksum_method_db_write: UniFFI API checksum mismatch") } @@ -997,7 +997,7 @@ func uniffiCheckChecksums() { checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { return C.uniffi_slatedb_uniffi_checksum_method_db_write_with_options() }) - if checksum != 13580 { + if checksum != 36986 { // If this happens try cleaning and rebuilding your project panic("slatedb: uniffi_slatedb_uniffi_checksum_method_db_write_with_options: UniFFI API checksum mismatch") } @@ -1186,7 +1186,7 @@ func uniffiCheckChecksums() { checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { return C.uniffi_slatedb_uniffi_checksum_method_dbtransaction_commit() }) - if checksum != 56467 { + if checksum != 56426 { // If this happens try cleaning and rebuilding your project panic("slatedb: uniffi_slatedb_uniffi_checksum_method_dbtransaction_commit: UniFFI API checksum mismatch") } @@ -1195,7 +1195,7 @@ func uniffiCheckChecksums() { checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { return C.uniffi_slatedb_uniffi_checksum_method_dbtransaction_commit_with_options() }) - if checksum != 62589 { + if checksum != 5743 { // If this happens try cleaning and rebuilding your project panic("slatedb: uniffi_slatedb_uniffi_checksum_method_dbtransaction_commit_with_options: UniFFI API checksum mismatch") } @@ -1704,6 +1704,33 @@ func uniffiCheckChecksums() { panic("slatedb: uniffi_slatedb_uniffi_checksum_method_writebatch_put_with_options: UniFFI API checksum mismatch") } } + { + checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { + return C.uniffi_slatedb_uniffi_checksum_method_writehandle_await_durable() + }) + if checksum != 2953 { + // If this happens try cleaning and rebuilding your project + panic("slatedb: uniffi_slatedb_uniffi_checksum_method_writehandle_await_durable: UniFFI API checksum mismatch") + } + } + { + checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { + return C.uniffi_slatedb_uniffi_checksum_method_writehandle_create_ts() + }) + if checksum != 16841 { + // If this happens try cleaning and rebuilding your project + panic("slatedb: uniffi_slatedb_uniffi_checksum_method_writehandle_create_ts: UniFFI API checksum mismatch") + } + } + { + checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { + return C.uniffi_slatedb_uniffi_checksum_method_writehandle_seqnum() + }) + if checksum != 19654 { + // If this happens try cleaning and rebuilding your project + panic("slatedb: uniffi_slatedb_uniffi_checksum_method_writehandle_seqnum: UniFFI API checksum mismatch") + } + } { checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { return C.uniffi_slatedb_uniffi_checksum_constructor_adminbuilder_new() @@ -3265,9 +3292,9 @@ type DbInterface interface { // Starts a transaction at the requested isolation level. Begin(isolationLevel IsolationLevel) (*DbTransaction, error) // Deletes `key` and returns metadata for the write. - Delete(key []byte) (WriteHandle, error) + Delete(key []byte) (*WriteHandle, error) // Deletes `key` using custom write options. - DeleteWithOptions(key []byte, options WriteOptions) (WriteHandle, error) + DeleteWithOptions(key []byte, options WriteOptions) (*WriteHandle, error) // Best-effort eviction of block-cache entries for one SST. // // If no block cache is configured, returns `Ok(())`. @@ -3285,16 +3312,16 @@ type DbInterface interface { // Reads the current value for `key` using custom read options. GetWithOptions(key []byte, options ReadOptions) (*[]byte, error) // Appends a merge operand for `key` and returns metadata for the write. - Merge(key []byte, operand []byte) (WriteHandle, error) + Merge(key []byte, operand []byte) (*WriteHandle, error) // Appends a merge operand using custom merge and write options. - MergeWithOptions(key []byte, operand []byte, mergeOptions MergeOptions, writeOptions WriteOptions) (WriteHandle, error) + MergeWithOptions(key []byte, operand []byte, mergeOptions MergeOptions, writeOptions WriteOptions) (*WriteHandle, error) // Inserts or overwrites a value and returns metadata for the write. // // Keys must be non-empty and at most `u16::MAX` bytes. Values must be at // most `u32::MAX` bytes. - Put(key []byte, value []byte) (WriteHandle, error) + Put(key []byte, value []byte) (*WriteHandle, error) // Inserts or overwrites a value using custom put and write options. - PutWithOptions(key []byte, value []byte, putOptions PutOptions, writeOptions WriteOptions) (WriteHandle, error) + PutWithOptions(key []byte, value []byte, putOptions PutOptions, writeOptions WriteOptions) (*WriteHandle, error) // Scans rows inside `range`. Scan(varRange KeyRange) (*DbIterator, error) // Scans rows whose keys start with `prefix`, restricted to `subrange`. @@ -3320,11 +3347,11 @@ type DbInterface interface { // Applies all operations in `batch` atomically. // // The provided batch is consumed and cannot be reused afterwards. - Write(batch *WriteBatch) (WriteHandle, error) + Write(batch *WriteBatch) (*WriteHandle, error) // Applies all operations in `batch` atomically using custom write options. // // The provided batch is consumed and cannot be reused afterwards. - WriteWithOptions(batch *WriteBatch, options WriteOptions) (WriteHandle, error) + WriteWithOptions(batch *WriteBatch, options WriteOptions) (*WriteHandle, error) } // A writable SlateDB handle. @@ -3367,31 +3394,29 @@ func (_self *Db) Begin(isolationLevel IsolationLevel) (*DbTransaction, error) { } // Deletes `key` and returns metadata for the write. -func (_self *Db) Delete(key []byte) (WriteHandle, error) { +func (_self *Db) Delete(key []byte) (*WriteHandle, error) { _pointer := _self.ffiObject.incrementPointer("*Db") defer _self.ffiObject.decrementPointer() res, err := uniffiRustCallAsync[*Error]( FfiConverterErrorINSTANCE, // completeFn - func(handle C.uint64_t, status *C.RustCallStatus) RustBufferI { - res := C.ffi_slatedb_uniffi_rust_future_complete_rust_buffer(handle, status) - return GoRustBuffer{ - inner: res, - } + func(handle C.uint64_t, status *C.RustCallStatus) C.uint64_t { + res := C.ffi_slatedb_uniffi_rust_future_complete_u64(handle, status) + return res }, // liftFn - func(ffi RustBufferI) WriteHandle { + func(ffi C.uint64_t) *WriteHandle { return FfiConverterWriteHandleINSTANCE.Lift(ffi) }, C.uniffi_slatedb_uniffi_fn_method_db_delete( _pointer, FfiConverterBytesINSTANCE.Lower(key)), // pollFn func(handle C.uint64_t, continuation C.UniffiRustFutureContinuationCallback, data C.uint64_t) { - C.ffi_slatedb_uniffi_rust_future_poll_rust_buffer(handle, continuation, data) + C.ffi_slatedb_uniffi_rust_future_poll_u64(handle, continuation, data) }, // freeFn func(handle C.uint64_t) { - C.ffi_slatedb_uniffi_rust_future_free_rust_buffer(handle) + C.ffi_slatedb_uniffi_rust_future_free_u64(handle) }, ) @@ -3403,31 +3428,29 @@ func (_self *Db) Delete(key []byte) (WriteHandle, error) { } // Deletes `key` using custom write options. -func (_self *Db) DeleteWithOptions(key []byte, options WriteOptions) (WriteHandle, error) { +func (_self *Db) DeleteWithOptions(key []byte, options WriteOptions) (*WriteHandle, error) { _pointer := _self.ffiObject.incrementPointer("*Db") defer _self.ffiObject.decrementPointer() res, err := uniffiRustCallAsync[*Error]( FfiConverterErrorINSTANCE, // completeFn - func(handle C.uint64_t, status *C.RustCallStatus) RustBufferI { - res := C.ffi_slatedb_uniffi_rust_future_complete_rust_buffer(handle, status) - return GoRustBuffer{ - inner: res, - } + func(handle C.uint64_t, status *C.RustCallStatus) C.uint64_t { + res := C.ffi_slatedb_uniffi_rust_future_complete_u64(handle, status) + return res }, // liftFn - func(ffi RustBufferI) WriteHandle { + func(ffi C.uint64_t) *WriteHandle { return FfiConverterWriteHandleINSTANCE.Lift(ffi) }, C.uniffi_slatedb_uniffi_fn_method_db_delete_with_options( _pointer, FfiConverterBytesINSTANCE.Lower(key), FfiConverterWriteOptionsINSTANCE.Lower(options)), // pollFn func(handle C.uint64_t, continuation C.UniffiRustFutureContinuationCallback, data C.uint64_t) { - C.ffi_slatedb_uniffi_rust_future_poll_rust_buffer(handle, continuation, data) + C.ffi_slatedb_uniffi_rust_future_poll_u64(handle, continuation, data) }, // freeFn func(handle C.uint64_t) { - C.ffi_slatedb_uniffi_rust_future_free_rust_buffer(handle) + C.ffi_slatedb_uniffi_rust_future_free_u64(handle) }, ) @@ -3681,31 +3704,29 @@ func (_self *Db) GetWithOptions(key []byte, options ReadOptions) (*[]byte, error } // Appends a merge operand for `key` and returns metadata for the write. -func (_self *Db) Merge(key []byte, operand []byte) (WriteHandle, error) { +func (_self *Db) Merge(key []byte, operand []byte) (*WriteHandle, error) { _pointer := _self.ffiObject.incrementPointer("*Db") defer _self.ffiObject.decrementPointer() res, err := uniffiRustCallAsync[*Error]( FfiConverterErrorINSTANCE, // completeFn - func(handle C.uint64_t, status *C.RustCallStatus) RustBufferI { - res := C.ffi_slatedb_uniffi_rust_future_complete_rust_buffer(handle, status) - return GoRustBuffer{ - inner: res, - } + func(handle C.uint64_t, status *C.RustCallStatus) C.uint64_t { + res := C.ffi_slatedb_uniffi_rust_future_complete_u64(handle, status) + return res }, // liftFn - func(ffi RustBufferI) WriteHandle { + func(ffi C.uint64_t) *WriteHandle { return FfiConverterWriteHandleINSTANCE.Lift(ffi) }, C.uniffi_slatedb_uniffi_fn_method_db_merge( _pointer, FfiConverterBytesINSTANCE.Lower(key), FfiConverterBytesINSTANCE.Lower(operand)), // pollFn func(handle C.uint64_t, continuation C.UniffiRustFutureContinuationCallback, data C.uint64_t) { - C.ffi_slatedb_uniffi_rust_future_poll_rust_buffer(handle, continuation, data) + C.ffi_slatedb_uniffi_rust_future_poll_u64(handle, continuation, data) }, // freeFn func(handle C.uint64_t) { - C.ffi_slatedb_uniffi_rust_future_free_rust_buffer(handle) + C.ffi_slatedb_uniffi_rust_future_free_u64(handle) }, ) @@ -3717,31 +3738,29 @@ func (_self *Db) Merge(key []byte, operand []byte) (WriteHandle, error) { } // Appends a merge operand using custom merge and write options. -func (_self *Db) MergeWithOptions(key []byte, operand []byte, mergeOptions MergeOptions, writeOptions WriteOptions) (WriteHandle, error) { +func (_self *Db) MergeWithOptions(key []byte, operand []byte, mergeOptions MergeOptions, writeOptions WriteOptions) (*WriteHandle, error) { _pointer := _self.ffiObject.incrementPointer("*Db") defer _self.ffiObject.decrementPointer() res, err := uniffiRustCallAsync[*Error]( FfiConverterErrorINSTANCE, // completeFn - func(handle C.uint64_t, status *C.RustCallStatus) RustBufferI { - res := C.ffi_slatedb_uniffi_rust_future_complete_rust_buffer(handle, status) - return GoRustBuffer{ - inner: res, - } + func(handle C.uint64_t, status *C.RustCallStatus) C.uint64_t { + res := C.ffi_slatedb_uniffi_rust_future_complete_u64(handle, status) + return res }, // liftFn - func(ffi RustBufferI) WriteHandle { + func(ffi C.uint64_t) *WriteHandle { return FfiConverterWriteHandleINSTANCE.Lift(ffi) }, C.uniffi_slatedb_uniffi_fn_method_db_merge_with_options( _pointer, FfiConverterBytesINSTANCE.Lower(key), FfiConverterBytesINSTANCE.Lower(operand), FfiConverterMergeOptionsINSTANCE.Lower(mergeOptions), FfiConverterWriteOptionsINSTANCE.Lower(writeOptions)), // pollFn func(handle C.uint64_t, continuation C.UniffiRustFutureContinuationCallback, data C.uint64_t) { - C.ffi_slatedb_uniffi_rust_future_poll_rust_buffer(handle, continuation, data) + C.ffi_slatedb_uniffi_rust_future_poll_u64(handle, continuation, data) }, // freeFn func(handle C.uint64_t) { - C.ffi_slatedb_uniffi_rust_future_free_rust_buffer(handle) + C.ffi_slatedb_uniffi_rust_future_free_u64(handle) }, ) @@ -3756,31 +3775,29 @@ func (_self *Db) MergeWithOptions(key []byte, operand []byte, mergeOptions Merge // // Keys must be non-empty and at most `u16::MAX` bytes. Values must be at // most `u32::MAX` bytes. -func (_self *Db) Put(key []byte, value []byte) (WriteHandle, error) { +func (_self *Db) Put(key []byte, value []byte) (*WriteHandle, error) { _pointer := _self.ffiObject.incrementPointer("*Db") defer _self.ffiObject.decrementPointer() res, err := uniffiRustCallAsync[*Error]( FfiConverterErrorINSTANCE, // completeFn - func(handle C.uint64_t, status *C.RustCallStatus) RustBufferI { - res := C.ffi_slatedb_uniffi_rust_future_complete_rust_buffer(handle, status) - return GoRustBuffer{ - inner: res, - } + func(handle C.uint64_t, status *C.RustCallStatus) C.uint64_t { + res := C.ffi_slatedb_uniffi_rust_future_complete_u64(handle, status) + return res }, // liftFn - func(ffi RustBufferI) WriteHandle { + func(ffi C.uint64_t) *WriteHandle { return FfiConverterWriteHandleINSTANCE.Lift(ffi) }, C.uniffi_slatedb_uniffi_fn_method_db_put( _pointer, FfiConverterBytesINSTANCE.Lower(key), FfiConverterBytesINSTANCE.Lower(value)), // pollFn func(handle C.uint64_t, continuation C.UniffiRustFutureContinuationCallback, data C.uint64_t) { - C.ffi_slatedb_uniffi_rust_future_poll_rust_buffer(handle, continuation, data) + C.ffi_slatedb_uniffi_rust_future_poll_u64(handle, continuation, data) }, // freeFn func(handle C.uint64_t) { - C.ffi_slatedb_uniffi_rust_future_free_rust_buffer(handle) + C.ffi_slatedb_uniffi_rust_future_free_u64(handle) }, ) @@ -3792,31 +3809,29 @@ func (_self *Db) Put(key []byte, value []byte) (WriteHandle, error) { } // Inserts or overwrites a value using custom put and write options. -func (_self *Db) PutWithOptions(key []byte, value []byte, putOptions PutOptions, writeOptions WriteOptions) (WriteHandle, error) { +func (_self *Db) PutWithOptions(key []byte, value []byte, putOptions PutOptions, writeOptions WriteOptions) (*WriteHandle, error) { _pointer := _self.ffiObject.incrementPointer("*Db") defer _self.ffiObject.decrementPointer() res, err := uniffiRustCallAsync[*Error]( FfiConverterErrorINSTANCE, // completeFn - func(handle C.uint64_t, status *C.RustCallStatus) RustBufferI { - res := C.ffi_slatedb_uniffi_rust_future_complete_rust_buffer(handle, status) - return GoRustBuffer{ - inner: res, - } + func(handle C.uint64_t, status *C.RustCallStatus) C.uint64_t { + res := C.ffi_slatedb_uniffi_rust_future_complete_u64(handle, status) + return res }, // liftFn - func(ffi RustBufferI) WriteHandle { + func(ffi C.uint64_t) *WriteHandle { return FfiConverterWriteHandleINSTANCE.Lift(ffi) }, C.uniffi_slatedb_uniffi_fn_method_db_put_with_options( _pointer, FfiConverterBytesINSTANCE.Lower(key), FfiConverterBytesINSTANCE.Lower(value), FfiConverterPutOptionsINSTANCE.Lower(putOptions), FfiConverterWriteOptionsINSTANCE.Lower(writeOptions)), // pollFn func(handle C.uint64_t, continuation C.UniffiRustFutureContinuationCallback, data C.uint64_t) { - C.ffi_slatedb_uniffi_rust_future_poll_rust_buffer(handle, continuation, data) + C.ffi_slatedb_uniffi_rust_future_poll_u64(handle, continuation, data) }, // freeFn func(handle C.uint64_t) { - C.ffi_slatedb_uniffi_rust_future_free_rust_buffer(handle) + C.ffi_slatedb_uniffi_rust_future_free_u64(handle) }, ) @@ -4082,31 +4097,29 @@ func (_self *Db) WarmSst(sstId SsTableId, targets []CacheTarget) error { // Applies all operations in `batch` atomically. // // The provided batch is consumed and cannot be reused afterwards. -func (_self *Db) Write(batch *WriteBatch) (WriteHandle, error) { +func (_self *Db) Write(batch *WriteBatch) (*WriteHandle, error) { _pointer := _self.ffiObject.incrementPointer("*Db") defer _self.ffiObject.decrementPointer() res, err := uniffiRustCallAsync[*Error]( FfiConverterErrorINSTANCE, // completeFn - func(handle C.uint64_t, status *C.RustCallStatus) RustBufferI { - res := C.ffi_slatedb_uniffi_rust_future_complete_rust_buffer(handle, status) - return GoRustBuffer{ - inner: res, - } + func(handle C.uint64_t, status *C.RustCallStatus) C.uint64_t { + res := C.ffi_slatedb_uniffi_rust_future_complete_u64(handle, status) + return res }, // liftFn - func(ffi RustBufferI) WriteHandle { + func(ffi C.uint64_t) *WriteHandle { return FfiConverterWriteHandleINSTANCE.Lift(ffi) }, C.uniffi_slatedb_uniffi_fn_method_db_write( _pointer, FfiConverterWriteBatchINSTANCE.Lower(batch)), // pollFn func(handle C.uint64_t, continuation C.UniffiRustFutureContinuationCallback, data C.uint64_t) { - C.ffi_slatedb_uniffi_rust_future_poll_rust_buffer(handle, continuation, data) + C.ffi_slatedb_uniffi_rust_future_poll_u64(handle, continuation, data) }, // freeFn func(handle C.uint64_t) { - C.ffi_slatedb_uniffi_rust_future_free_rust_buffer(handle) + C.ffi_slatedb_uniffi_rust_future_free_u64(handle) }, ) @@ -4120,31 +4133,29 @@ func (_self *Db) Write(batch *WriteBatch) (WriteHandle, error) { // Applies all operations in `batch` atomically using custom write options. // // The provided batch is consumed and cannot be reused afterwards. -func (_self *Db) WriteWithOptions(batch *WriteBatch, options WriteOptions) (WriteHandle, error) { +func (_self *Db) WriteWithOptions(batch *WriteBatch, options WriteOptions) (*WriteHandle, error) { _pointer := _self.ffiObject.incrementPointer("*Db") defer _self.ffiObject.decrementPointer() res, err := uniffiRustCallAsync[*Error]( FfiConverterErrorINSTANCE, // completeFn - func(handle C.uint64_t, status *C.RustCallStatus) RustBufferI { - res := C.ffi_slatedb_uniffi_rust_future_complete_rust_buffer(handle, status) - return GoRustBuffer{ - inner: res, - } + func(handle C.uint64_t, status *C.RustCallStatus) C.uint64_t { + res := C.ffi_slatedb_uniffi_rust_future_complete_u64(handle, status) + return res }, // liftFn - func(ffi RustBufferI) WriteHandle { + func(ffi C.uint64_t) *WriteHandle { return FfiConverterWriteHandleINSTANCE.Lift(ffi) }, C.uniffi_slatedb_uniffi_fn_method_db_write_with_options( _pointer, FfiConverterWriteBatchINSTANCE.Lower(batch), FfiConverterWriteOptionsINSTANCE.Lower(options)), // pollFn func(handle C.uint64_t, continuation C.UniffiRustFutureContinuationCallback, data C.uint64_t) { - C.ffi_slatedb_uniffi_rust_future_poll_rust_buffer(handle, continuation, data) + C.ffi_slatedb_uniffi_rust_future_poll_u64(handle, continuation, data) }, // freeFn func(handle C.uint64_t) { - C.ffi_slatedb_uniffi_rust_future_free_rust_buffer(handle) + C.ffi_slatedb_uniffi_rust_future_free_u64(handle) }, ) @@ -5856,11 +5867,11 @@ type DbTransactionInterface interface { // Commits the transaction. // // Returns `None` when the transaction performed no writes. - Commit() (*WriteHandle, error) + Commit() (**WriteHandle, error) // Commits the transaction using custom write options. // // Returns `None` when the transaction performed no writes. - CommitWithOptions(options WriteOptions) (*WriteHandle, error) + CommitWithOptions(options WriteOptions) (**WriteHandle, error) // Buffers a delete inside the transaction. Delete(key []byte) error // Reads the value visible to this transaction for `key`. @@ -5912,7 +5923,7 @@ type DbTransaction struct { // Commits the transaction. // // Returns `None` when the transaction performed no writes. -func (_self *DbTransaction) Commit() (*WriteHandle, error) { +func (_self *DbTransaction) Commit() (**WriteHandle, error) { _pointer := _self.ffiObject.incrementPointer("*DbTransaction") defer _self.ffiObject.decrementPointer() res, err := uniffiRustCallAsync[*Error]( @@ -5925,7 +5936,7 @@ func (_self *DbTransaction) Commit() (*WriteHandle, error) { } }, // liftFn - func(ffi RustBufferI) *WriteHandle { + func(ffi RustBufferI) **WriteHandle { return FfiConverterOptionalWriteHandleINSTANCE.Lift(ffi) }, C.uniffi_slatedb_uniffi_fn_method_dbtransaction_commit( @@ -5950,7 +5961,7 @@ func (_self *DbTransaction) Commit() (*WriteHandle, error) { // Commits the transaction using custom write options. // // Returns `None` when the transaction performed no writes. -func (_self *DbTransaction) CommitWithOptions(options WriteOptions) (*WriteHandle, error) { +func (_self *DbTransaction) CommitWithOptions(options WriteOptions) (**WriteHandle, error) { _pointer := _self.ffiObject.incrementPointer("*DbTransaction") defer _self.ffiObject.decrementPointer() res, err := uniffiRustCallAsync[*Error]( @@ -5963,7 +5974,7 @@ func (_self *DbTransaction) CommitWithOptions(options WriteOptions) (*WriteHandl } }, // liftFn - func(ffi RustBufferI) *WriteHandle { + func(ffi RustBufferI) **WriteHandle { return FfiConverterOptionalWriteHandleINSTANCE.Lift(ffi) }, C.uniffi_slatedb_uniffi_fn_method_dbtransaction_commit_with_options( @@ -8876,6 +8887,128 @@ func (_ FfiDestroyerWriteBatch) Destroy(value *WriteBatch) { value.Destroy() } +// Handle returned by a successful write. +type WriteHandleInterface interface { + // Waits until the write has been durably persisted. + AwaitDurable() error + // Returns the creation timestamp assigned to the write. + CreateTs() int64 + // Returns the sequence number assigned to the write. + Seqnum() uint64 +} + +// Handle returned by a successful write. +type WriteHandle struct { + ffiObject FfiObject +} + +// Waits until the write has been durably persisted. +func (_self *WriteHandle) AwaitDurable() error { + _pointer := _self.ffiObject.incrementPointer("*WriteHandle") + defer _self.ffiObject.decrementPointer() + _, err := uniffiRustCallAsync[*Error]( + FfiConverterErrorINSTANCE, + // completeFn + func(handle C.uint64_t, status *C.RustCallStatus) struct{} { + C.ffi_slatedb_uniffi_rust_future_complete_void(handle, status) + return struct{}{} + }, + // liftFn + func(_ struct{}) struct{} { return struct{}{} }, + C.uniffi_slatedb_uniffi_fn_method_writehandle_await_durable( + _pointer), + // pollFn + func(handle C.uint64_t, continuation C.UniffiRustFutureContinuationCallback, data C.uint64_t) { + C.ffi_slatedb_uniffi_rust_future_poll_void(handle, continuation, data) + }, + // freeFn + func(handle C.uint64_t) { + C.ffi_slatedb_uniffi_rust_future_free_void(handle) + }, + ) + + if err == nil { + return nil + } + + return err +} + +// Returns the creation timestamp assigned to the write. +func (_self *WriteHandle) CreateTs() int64 { + _pointer := _self.ffiObject.incrementPointer("*WriteHandle") + defer _self.ffiObject.decrementPointer() + return FfiConverterInt64INSTANCE.Lift(rustCall(func(_uniffiStatus *C.RustCallStatus) C.int64_t { + return C.uniffi_slatedb_uniffi_fn_method_writehandle_create_ts( + _pointer, _uniffiStatus) + })) +} + +// Returns the sequence number assigned to the write. +func (_self *WriteHandle) Seqnum() uint64 { + _pointer := _self.ffiObject.incrementPointer("*WriteHandle") + defer _self.ffiObject.decrementPointer() + return FfiConverterUint64INSTANCE.Lift(rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint64_t { + return C.uniffi_slatedb_uniffi_fn_method_writehandle_seqnum( + _pointer, _uniffiStatus) + })) +} +func (object *WriteHandle) Destroy() { + runtime.SetFinalizer(object, nil) + object.ffiObject.destroy() +} + +type FfiConverterWriteHandle struct{} + +var FfiConverterWriteHandleINSTANCE = FfiConverterWriteHandle{} + +func (c FfiConverterWriteHandle) Lift(handle C.uint64_t) *WriteHandle { + result := &WriteHandle{ + newFfiObject( + handle, + func(handle C.uint64_t, status *C.RustCallStatus) C.uint64_t { + return C.uniffi_slatedb_uniffi_fn_clone_writehandle(handle, status) + }, + func(handle C.uint64_t, status *C.RustCallStatus) { + C.uniffi_slatedb_uniffi_fn_free_writehandle(handle, status) + }, + ), + } + runtime.SetFinalizer(result, (*WriteHandle).Destroy) + return result +} + +func (c FfiConverterWriteHandle) Read(reader io.Reader) *WriteHandle { + return c.Lift(C.uint64_t(readUint64(reader))) +} + +func (c FfiConverterWriteHandle) Lower(value *WriteHandle) C.uint64_t { + // TODO: this is bad - all synchronization from ObjectRuntime.go is discarded here, + // because the handle will be decremented immediately after this function returns, + // and someone will be left holding onto a non-locked handle. + handle := value.ffiObject.incrementPointer("*WriteHandle") + defer value.ffiObject.decrementPointer() + return handle +} + +func (c FfiConverterWriteHandle) Write(writer io.Writer, value *WriteHandle) { + writeUint64(writer, uint64(c.Lower(value))) +} + +func LiftFromExternalWriteHandle(handle uint64) *WriteHandle { + return FfiConverterWriteHandleINSTANCE.Lift(C.uint64_t(handle)) +} + +func LowerToExternalWriteHandle(value *WriteHandle) uint64 { + return uint64(FfiConverterWriteHandleINSTANCE.Lower(value)) +} + +type FfiDestroyerWriteHandle struct{} + +func (_ FfiDestroyerWriteHandle) Destroy(value *WriteHandle) { + value.Destroy() +} + // Options controlling how a bloom filter policy is constructed. // // Pass an optional prefix extractor as a separate constructor parameter; it @@ -11039,61 +11172,14 @@ func (_ FfiDestroyerVersionedManifest) Destroy(value VersionedManifest) { value.Destroy() } -// Metadata returned by a successful write. -type WriteHandle struct { - // Sequence number assigned to the write. - Seqnum uint64 - // Creation timestamp assigned to the write. - CreateTs int64 -} - -func (r *WriteHandle) Destroy() { - FfiDestroyerUint64{}.Destroy(r.Seqnum) - FfiDestroyerInt64{}.Destroy(r.CreateTs) -} - -type FfiConverterWriteHandle struct{} - -var FfiConverterWriteHandleINSTANCE = FfiConverterWriteHandle{} - -func (c FfiConverterWriteHandle) Lift(rb RustBufferI) WriteHandle { - return LiftFromRustBuffer[WriteHandle](c, rb) -} - -func (c FfiConverterWriteHandle) Read(reader io.Reader) WriteHandle { - return WriteHandle{ - FfiConverterUint64INSTANCE.Read(reader), - FfiConverterInt64INSTANCE.Read(reader), - } -} - -func (c FfiConverterWriteHandle) Lower(value WriteHandle) C.RustBuffer { - return LowerIntoRustBuffer[WriteHandle](c, value) -} - -func (c FfiConverterWriteHandle) LowerExternal(value WriteHandle) ExternalCRustBuffer { - return RustBufferFromC(LowerIntoRustBuffer[WriteHandle](c, value)) -} - -func (c FfiConverterWriteHandle) Write(writer io.Writer, value WriteHandle) { - FfiConverterUint64INSTANCE.Write(writer, value.Seqnum) - FfiConverterInt64INSTANCE.Write(writer, value.CreateTs) -} - -type FfiDestroyerWriteHandle struct{} - -func (_ FfiDestroyerWriteHandle) Destroy(value WriteHandle) { - value.Destroy() -} - -// Options that control durability behavior for writes and commits. +// Options that control writes and commits. type WriteOptions struct { - // Whether the call waits for the write to become durable before returning. - AwaitDurable bool + // Optional caller-supplied sequence number. Zero uses SlateDB's sequence oracle. + Seqnum uint64 } func (r *WriteOptions) Destroy() { - FfiDestroyerBool{}.Destroy(r.AwaitDurable) + FfiDestroyerUint64{}.Destroy(r.Seqnum) } type FfiConverterWriteOptions struct{} @@ -11106,7 +11192,7 @@ func (c FfiConverterWriteOptions) Lift(rb RustBufferI) WriteOptions { func (c FfiConverterWriteOptions) Read(reader io.Reader) WriteOptions { return WriteOptions{ - FfiConverterBoolINSTANCE.Read(reader), + FfiConverterUint64INSTANCE.Read(reader), } } @@ -11119,7 +11205,7 @@ func (c FfiConverterWriteOptions) LowerExternal(value WriteOptions) ExternalCRus } func (c FfiConverterWriteOptions) Write(writer io.Writer, value WriteOptions) { - FfiConverterBoolINSTANCE.Write(writer, value.AwaitDurable) + FfiConverterUint64INSTANCE.Write(writer, value.Seqnum) } type FfiDestroyerWriteOptions struct{} @@ -13078,6 +13164,47 @@ func (_ FfiDestroyerOptionalPrefixExtractor) Destroy(value *PrefixExtractor) { } } +type FfiConverterOptionalWriteHandle struct{} + +var FfiConverterOptionalWriteHandleINSTANCE = FfiConverterOptionalWriteHandle{} + +func (c FfiConverterOptionalWriteHandle) Lift(rb RustBufferI) **WriteHandle { + return LiftFromRustBuffer[**WriteHandle](c, rb) +} + +func (_ FfiConverterOptionalWriteHandle) Read(reader io.Reader) **WriteHandle { + if readInt8(reader) == 0 { + return nil + } + temp := FfiConverterWriteHandleINSTANCE.Read(reader) + return &temp +} + +func (c FfiConverterOptionalWriteHandle) Lower(value **WriteHandle) C.RustBuffer { + return LowerIntoRustBuffer[**WriteHandle](c, value) +} + +func (c FfiConverterOptionalWriteHandle) LowerExternal(value **WriteHandle) ExternalCRustBuffer { + return RustBufferFromC(LowerIntoRustBuffer[**WriteHandle](c, value)) +} + +func (_ FfiConverterOptionalWriteHandle) Write(writer io.Writer, value **WriteHandle) { + if value == nil { + writeInt8(writer, 0) + } else { + writeInt8(writer, 1) + FfiConverterWriteHandleINSTANCE.Write(writer, *value) + } +} + +type FfiDestroyerOptionalWriteHandle struct{} + +func (_ FfiDestroyerOptionalWriteHandle) Destroy(value **WriteHandle) { + if value != nil { + FfiDestroyerWriteHandle{}.Destroy(*value) + } +} + type FfiConverterOptionalCompaction struct{} var FfiConverterOptionalCompactionINSTANCE = FfiConverterOptionalCompaction{} @@ -13488,47 +13615,6 @@ func (_ FfiDestroyerOptionalVersionedManifest) Destroy(value *VersionedManifest) } } -type FfiConverterOptionalWriteHandle struct{} - -var FfiConverterOptionalWriteHandleINSTANCE = FfiConverterOptionalWriteHandle{} - -func (c FfiConverterOptionalWriteHandle) Lift(rb RustBufferI) *WriteHandle { - return LiftFromRustBuffer[*WriteHandle](c, rb) -} - -func (_ FfiConverterOptionalWriteHandle) Read(reader io.Reader) *WriteHandle { - if readInt8(reader) == 0 { - return nil - } - temp := FfiConverterWriteHandleINSTANCE.Read(reader) - return &temp -} - -func (c FfiConverterOptionalWriteHandle) Lower(value *WriteHandle) C.RustBuffer { - return LowerIntoRustBuffer[*WriteHandle](c, value) -} - -func (c FfiConverterOptionalWriteHandle) LowerExternal(value *WriteHandle) ExternalCRustBuffer { - return RustBufferFromC(LowerIntoRustBuffer[*WriteHandle](c, value)) -} - -func (_ FfiConverterOptionalWriteHandle) Write(writer io.Writer, value *WriteHandle) { - if value == nil { - writeInt8(writer, 0) - } else { - writeInt8(writer, 1) - FfiConverterWriteHandleINSTANCE.Write(writer, *value) - } -} - -type FfiDestroyerOptionalWriteHandle struct{} - -func (_ FfiDestroyerOptionalWriteHandle) Destroy(value *WriteHandle) { - if value != nil { - FfiDestroyerWriteHandle{}.Destroy(*value) - } -} - type FfiConverterOptionalCloseReason struct{} var FfiConverterOptionalCloseReasonINSTANCE = FfiConverterOptionalCloseReason{} diff --git a/bindings/go/uniffi/slatedb.h b/bindings/go/uniffi/slatedb.h index 5d21f81ceb..ed46683e11 100644 --- a/bindings/go/uniffi/slatedb.h +++ b/bindings/go/uniffi/slatedb.h @@ -1753,6 +1753,31 @@ void uniffi_slatedb_uniffi_fn_method_writebatch_put(uint64_t ptr, RustBuffer key void uniffi_slatedb_uniffi_fn_method_writebatch_put_with_options(uint64_t ptr, RustBuffer key, RustBuffer value, RustBuffer options, RustCallStatus *out_status ); #endif +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_CLONE_WRITEHANDLE +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_CLONE_WRITEHANDLE +uint64_t uniffi_slatedb_uniffi_fn_clone_writehandle(uint64_t handle, RustCallStatus *out_status +); +#endif +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_FREE_WRITEHANDLE +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_FREE_WRITEHANDLE +void uniffi_slatedb_uniffi_fn_free_writehandle(uint64_t handle, RustCallStatus *out_status +); +#endif +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_WRITEHANDLE_AWAIT_DURABLE +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_WRITEHANDLE_AWAIT_DURABLE +uint64_t uniffi_slatedb_uniffi_fn_method_writehandle_await_durable(uint64_t ptr +); +#endif +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_WRITEHANDLE_CREATE_TS +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_WRITEHANDLE_CREATE_TS +int64_t uniffi_slatedb_uniffi_fn_method_writehandle_create_ts(uint64_t ptr, RustCallStatus *out_status +); +#endif +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_WRITEHANDLE_SEQNUM +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_WRITEHANDLE_SEQNUM +uint64_t uniffi_slatedb_uniffi_fn_method_writehandle_seqnum(uint64_t ptr, RustCallStatus *out_status +); +#endif #ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_FUNC_INIT_LOGGING #define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_FUNC_INIT_LOGGING void uniffi_slatedb_uniffi_fn_func_init_logging(RustBuffer level, RustBuffer callback, RustCallStatus *out_status @@ -2898,6 +2923,24 @@ uint16_t uniffi_slatedb_uniffi_checksum_method_writebatch_put(void #define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_WRITEBATCH_PUT_WITH_OPTIONS uint16_t uniffi_slatedb_uniffi_checksum_method_writebatch_put_with_options(void +); +#endif +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_WRITEHANDLE_AWAIT_DURABLE +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_WRITEHANDLE_AWAIT_DURABLE +uint16_t uniffi_slatedb_uniffi_checksum_method_writehandle_await_durable(void + +); +#endif +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_WRITEHANDLE_CREATE_TS +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_WRITEHANDLE_CREATE_TS +uint16_t uniffi_slatedb_uniffi_checksum_method_writehandle_create_ts(void + +); +#endif +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_WRITEHANDLE_SEQNUM +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_WRITEHANDLE_SEQNUM +uint16_t uniffi_slatedb_uniffi_checksum_method_writehandle_seqnum(void + ); #endif #ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_CONSTRUCTOR_ADMINBUILDER_NEW diff --git a/bindings/go/uniffi/slatedb_test.go b/bindings/go/uniffi/slatedb_test.go index ab07730c4b..5facf2507c 100644 --- a/bindings/go/uniffi/slatedb_test.go +++ b/bindings/go/uniffi/slatedb_test.go @@ -158,6 +158,22 @@ func uint64Ptr(v uint64) *uint64 { return &v } +func trackWriteHandle(t *testing.T, handle *slatedb.WriteHandle) *slatedb.WriteHandle { + t.Helper() + if handle == nil { + t.Fatal("got nil write handle") + } + t.Cleanup(handle.Destroy) + return handle +} + +func awaitDurable(t *testing.T, handle *slatedb.WriteHandle) { + t.Helper() + if err := handle.AwaitDurable(); err != nil { + t.Fatalf("WriteHandle.AwaitDurable(): %v", err) + } +} + func drainIterator(t *testing.T, iter *slatedb.DbIterator) []slatedb.KeyValue { t.Helper() @@ -597,16 +613,17 @@ func TestDbCrudAndMetadata(t *testing.T) { } putOptions := slatedb.PutOptions{Ttl: slatedb.TtlDefault{}} - writeOptions := slatedb.WriteOptions{AwaitDurable: true} + writeOptions := slatedb.WriteOptions{Seqnum: 0} firstWrite, err := handle.db.Put([]byte("alpha"), []byte("one")) if err != nil { t.Fatalf("Put(alpha): %v", err) } - if firstWrite.Seqnum == 0 { + firstWrite = trackWriteHandle(t, firstWrite) + if firstWrite.Seqnum() == 0 { t.Fatalf("Put(alpha): Seqnum = 0") } - if firstWrite.CreateTs == 0 { + if firstWrite.CreateTs() == 0 { t.Fatalf("Put(alpha): CreateTs = 0") } @@ -639,11 +656,11 @@ func TestDbCrudAndMetadata(t *testing.T) { if !bytes.Equal(metadata.Value, []byte("one")) { t.Fatalf("GetKeyValue(alpha): value = %q, want %q", metadata.Value, "one") } - if metadata.Seq != firstWrite.Seqnum { - t.Fatalf("GetKeyValue(alpha): seq = %d, want %d", metadata.Seq, firstWrite.Seqnum) + if metadata.Seq != firstWrite.Seqnum() { + t.Fatalf("GetKeyValue(alpha): seq = %d, want %d", metadata.Seq, firstWrite.Seqnum()) } - if metadata.CreateTs != firstWrite.CreateTs { - t.Fatalf("GetKeyValue(alpha): create ts = %d, want %d", metadata.CreateTs, firstWrite.CreateTs) + if metadata.CreateTs != firstWrite.CreateTs() { + t.Fatalf("GetKeyValue(alpha): create ts = %d, want %d", metadata.CreateTs, firstWrite.CreateTs()) } metadata, err = handle.db.GetKeyValueWithOptions([]byte("alpha"), readOptions) @@ -658,10 +675,12 @@ func TestDbCrudAndMetadata(t *testing.T) { if err != nil { t.Fatalf("PutWithOptions(beta): %v", err) } - if secondWrite.Seqnum <= firstWrite.Seqnum { - t.Fatalf("PutWithOptions(beta): seq = %d, want > %d", secondWrite.Seqnum, firstWrite.Seqnum) + secondWrite = trackWriteHandle(t, secondWrite) + awaitDurable(t, secondWrite) + if secondWrite.Seqnum() <= firstWrite.Seqnum() { + t.Fatalf("PutWithOptions(beta): seq = %d, want > %d", secondWrite.Seqnum(), firstWrite.Seqnum()) } - if secondWrite.CreateTs == 0 { + if secondWrite.CreateTs() == 0 { t.Fatalf("PutWithOptions(beta): CreateTs = 0") } @@ -696,8 +715,9 @@ func TestDbCrudAndMetadata(t *testing.T) { if err != nil { t.Fatalf("Delete(alpha): %v", err) } - if deleteWrite.Seqnum <= secondWrite.Seqnum { - t.Fatalf("Delete(alpha): seq = %d, want > %d", deleteWrite.Seqnum, secondWrite.Seqnum) + deleteWrite = trackWriteHandle(t, deleteWrite) + if deleteWrite.Seqnum() <= secondWrite.Seqnum() { + t.Fatalf("Delete(alpha): seq = %d, want > %d", deleteWrite.Seqnum(), secondWrite.Seqnum()) } value, err = handle.db.Get([]byte("alpha")) @@ -712,8 +732,9 @@ func TestDbCrudAndMetadata(t *testing.T) { if err != nil { t.Fatalf("DeleteWithOptions(beta): %v", err) } - if deleteWrite.Seqnum <= secondWrite.Seqnum { - t.Fatalf("DeleteWithOptions(beta): seq = %d, want > %d", deleteWrite.Seqnum, secondWrite.Seqnum) + deleteWrite = trackWriteHandle(t, deleteWrite) + if deleteWrite.Seqnum() <= secondWrite.Seqnum() { + t.Fatalf("DeleteWithOptions(beta): seq = %d, want > %d", deleteWrite.Seqnum(), secondWrite.Seqnum()) } value, err = handle.db.Get([]byte("beta")) @@ -846,7 +867,8 @@ func TestDbBatchWriteAndConsumption(t *testing.T) { if err != nil { t.Fatalf("Write(): %v", err) } - if batchWrite.Seqnum == 0 { + batchWrite = trackWriteHandle(t, batchWrite) + if batchWrite.Seqnum() == 0 { t.Fatalf("Write(): Seqnum = 0") } @@ -877,9 +899,12 @@ func TestDbBatchWriteAndConsumption(t *testing.T) { t.Fatalf("WriteBatch.PutWithOptions(): %v", err) } - if _, err := handle.db.WriteWithOptions(secondBatch, slatedb.WriteOptions{AwaitDurable: true}); err != nil { + secondBatchWrite, err := handle.db.WriteWithOptions(secondBatch, slatedb.WriteOptions{Seqnum: 0}) + if err != nil { t.Fatalf("WriteWithOptions(): %v", err) } + secondBatchWrite = trackWriteHandle(t, secondBatchWrite) + awaitDurable(t, secondBatchWrite) value, err = handle.db.Get([]byte("batch-put-2")) if err != nil { @@ -937,14 +962,17 @@ func TestDbMerge(t *testing.T) { t.Fatalf("Get(merge) after Merge(): got %v, want %q", value, "base:one") } - if _, err := handle.db.MergeWithOptions( + mergeWrite, err := handle.db.MergeWithOptions( []byte("merge"), []byte(":two"), slatedb.MergeOptions{Ttl: slatedb.TtlDefault{}}, - slatedb.WriteOptions{AwaitDurable: true}, - ); err != nil { + slatedb.WriteOptions{Seqnum: 0}, + ) + if err != nil { t.Fatalf("MergeWithOptions(): %v", err) } + mergeWrite = trackWriteHandle(t, mergeWrite) + awaitDurable(t, mergeWrite) value, err = handle.db.Get([]byte("merge")) if err != nil { @@ -1024,11 +1052,15 @@ func TestDbTransactions(t *testing.T) { t.Fatalf("db.Get(txn-key) before commit: got %q, want nil", *liveValue) } - commitHandle, err := tx.Commit() + optionalCommitHandle, err := tx.Commit() if err != nil { t.Fatalf("tx.Commit(): %v", err) } - if commitHandle == nil || commitHandle.Seqnum == 0 { + if optionalCommitHandle == nil { + t.Fatal("tx.Commit(): got nil write handle") + } + commitHandle := trackWriteHandle(t, *optionalCommitHandle) + if commitHandle.Seqnum() == 0 { t.Fatalf("tx.Commit(): got %v, want non-nil write handle", commitHandle) } @@ -1124,18 +1156,22 @@ func TestDbInvalidInputsAndErrorMapping(t *testing.T) { t.Fatalf("secondary Put(): %v", err) } - _, err := primary.db.Put([]byte("stale"), []byte("value")) + write, err := primary.db.Put([]byte("stale"), []byte("value")) + if err == nil { + write = trackWriteHandle(t, write) + err = write.AwaitDurable() + } if !errors.Is(err, slatedb.ErrErrorClosed) { - t.Fatalf("primary Put() after fencing: got %v, want closed error", err) + t.Fatalf("primary write durability after fencing: got %v, want closed error", err) } primary.open = false var closedErr *slatedb.ErrorClosed if !errors.As(err, &closedErr) { - t.Fatalf("primary Put() after fencing: expected *ErrorClosed, got %T", err) + t.Fatalf("primary write durability after fencing: expected *ErrorClosed, got %T", err) } if closedErr.Reason != slatedb.CloseReasonFenced { - t.Fatalf("primary Put() after fencing: got close reason %v, want %v", closedErr.Reason, slatedb.CloseReasonFenced) + t.Fatalf("primary write durability after fencing: got close reason %v, want %v", closedErr.Reason, slatedb.CloseReasonFenced) } }) } @@ -3073,11 +3109,13 @@ func TestDbTtl(t *testing.T) { key, value := []byte("alpha"), []byte("one") putOptions := slatedb.PutOptions{Ttl: slatedb.TtlExpireAtMillis{Field0: 1}} - writeOptions := slatedb.WriteOptions{AwaitDurable: true} - _, err := handle.db.PutWithOptions(key, value, putOptions, writeOptions) + writeOptions := slatedb.WriteOptions{Seqnum: 0} + write, err := handle.db.PutWithOptions(key, value, putOptions, writeOptions) if err != nil { t.Fatalf("Put(alpha): %v", err) } + write = trackWriteHandle(t, write) + awaitDurable(t, write) readerHandle := openTestReader(t, store, nil) @@ -3131,12 +3169,15 @@ const batchSeedTtlMillis = 3_600_000 func seedBatchRows(t *testing.T, db *slatedb.Db) { t.Helper() - writeOptions := slatedb.WriteOptions{AwaitDurable: true} + writeOptions := slatedb.WriteOptions{Seqnum: 0} for _, row := range batchSeedRows { putOptions := slatedb.PutOptions{Ttl: row.ttl} - if _, err := db.PutWithOptions([]byte(row.key), []byte(row.value), putOptions, writeOptions); err != nil { + write, err := db.PutWithOptions([]byte(row.key), []byte(row.value), putOptions, writeOptions) + if err != nil { t.Fatalf("PutWithOptions(%q): %v", row.key, err) } + write = trackWriteHandle(t, write) + awaitDurable(t, write) } } @@ -3387,7 +3428,7 @@ func openBenchDB(b *testing.B) *slatedb.Db { db.Destroy() }) - writeOptions := slatedb.WriteOptions{AwaitDurable: false} + writeOptions := slatedb.WriteOptions{Seqnum: 0} putOptions := slatedb.PutOptions{Ttl: slatedb.TtlDefault{}} for i := 0; i < benchScanRows; i++ { key := []byte(fmt.Sprintf("bench:%06d", i)) diff --git a/bindings/java/slatedb-uniffi/src/test/java/io/slatedb/uniffi/SlateDbDbTest.java b/bindings/java/slatedb-uniffi/src/test/java/io/slatedb/uniffi/SlateDbDbTest.java index 8ac8be369b..e226e67670 100644 --- a/bindings/java/slatedb-uniffi/src/test/java/io/slatedb/uniffi/SlateDbDbTest.java +++ b/bindings/java/slatedb-uniffi/src/test/java/io/slatedb/uniffi/SlateDbDbTest.java @@ -94,10 +94,11 @@ void dbCrudAndMetadata() throws Exception { ReadOptions readOptions = TestSupport.readOptions(); PutOptions putOptions = new PutOptions(new Ttl.Default()); - WriteOptions writeOptions = new WriteOptions(true); + WriteOptions writeOptions = new WriteOptions(0L); WriteHandle firstWrite = TestSupport.await(db.put(TestSupport.bytes("alpha"), TestSupport.bytes("one"))); assertNotNull(firstWrite); + TestSupport.await(firstWrite.awaitDurable()); assertTrue(firstWrite.seqnum() > 0); assertTrue(firstWrite.createTs() > 0); @@ -126,6 +127,7 @@ void dbCrudAndMetadata() throws Exception { putOptions, writeOptions)); assertNotNull(secondWrite); + TestSupport.await(secondWrite.awaitDurable()); assertTrue(secondWrite.seqnum() > firstWrite.seqnum()); assertTrue(secondWrite.createTs() > 0); @@ -265,7 +267,10 @@ void dbBatchWriteAndConsumption() throws Exception { TestSupport.bytes("value-2"), new PutOptions(new Ttl.Default())); - TestSupport.await(db.writeWithOptions(secondBatch, new WriteOptions(true))); + try (WriteHandle writeHandle = + TestSupport.await(db.writeWithOptions(secondBatch, new WriteOptions(0L)))) { + TestSupport.await(writeHandle.awaitDurable()); + } } assertArrayEquals( @@ -298,12 +303,15 @@ void dbMerge() throws Exception { TestSupport.await(db.merge(TestSupport.bytes("merge"), TestSupport.bytes(":one"))); assertArrayEquals(TestSupport.bytes("base:one"), TestSupport.await(db.get(TestSupport.bytes("merge")))); - TestSupport.await( - db.mergeWithOptions( - TestSupport.bytes("merge"), - TestSupport.bytes(":two"), - new MergeOptions(new Ttl.Default()), - new WriteOptions(true))); + try (WriteHandle writeHandle = + TestSupport.await( + db.mergeWithOptions( + TestSupport.bytes("merge"), + TestSupport.bytes(":two"), + new MergeOptions(new Ttl.Default()), + new WriteOptions(0L)))) { + TestSupport.await(writeHandle.awaitDurable()); + } assertArrayEquals( TestSupport.bytes("base:one:two"), TestSupport.await(db.get(TestSupport.bytes("merge")))); @@ -413,7 +421,15 @@ void dbWriterFencing() throws Exception { Error.Closed error = TestSupport.awaitFailure( Error.Closed.class, - primaryDb.put(TestSupport.bytes("stale"), TestSupport.bytes("value"))); + primaryDb + .put(TestSupport.bytes("stale"), TestSupport.bytes("value")) + .thenCompose( + writeHandle -> + writeHandle + .awaitDurable() + .whenComplete( + (ignored, failure) -> + writeHandle.close()))); primary.markClosed(); assertEquals(CloseReason.FENCED, error.reason()); diff --git a/bindings/node/tests/admin.test.mjs b/bindings/node/tests/admin.test.mjs index 08bf06a27e..8702ac05c7 100644 --- a/bindings/node/tests/admin.test.mjs +++ b/bindings/node/tests/admin.test.mjs @@ -72,26 +72,26 @@ test("admin manifest read list and state view", async (t) => { const admin = openAdmin(store, { path, cleanup }); const db = await openDb(store, { path, cleanup }); - const firstWrite = await db.put_with_options( + const firstWrite = cleanup.track(await db.put_with_options( bytes("alpha"), bytes("one"), putOptions(), writeOptions(false), - ); + )); await db.flush_with_options({ flush_type: FlushType.MemTable }); - const secondWrite = await db.put_with_options( + const secondWrite = cleanup.track(await db.put_with_options( bytes("beta"), bytes("two"), putOptions(), writeOptions(false), - ); + )); await db.flush_with_options({ flush_type: FlushType.MemTable }); const latest = await admin.read_manifest(undefined); assert.notEqual(latest, undefined); assert.ok(BigInt(latest.id) >= 3n); - assert.ok(BigInt(latest.last_l0_seq) >= BigInt(secondWrite.seqnum)); + assert.ok(BigInt(latest.last_l0_seq) >= BigInt(secondWrite.seqnum())); const first = await admin.read_manifest(1n); assert.notEqual(first, undefined); @@ -113,7 +113,7 @@ test("admin manifest read list and state view", async (t) => { const stateView = await admin.read_compactor_state_view(); assert.equal(BigInt(stateView.manifest.id), BigInt(latest.id)); - assert.ok(BigInt(firstWrite.seqnum) > 0n); + assert.ok(BigInt(firstWrite.seqnum()) > 0n); }); test("admin compaction queries handle empty store and invalid ids", async (t) => { @@ -234,27 +234,27 @@ test("admin sequence lookups use persisted tracker", async (t) => { const admin = openAdmin(store, { path, cleanup }); const db = await openDb(store, { path, cleanup }); - const firstWrite = await db.put_with_options( + const firstWrite = cleanup.track(await db.put_with_options( bytes("k1"), bytes("v1"), putOptions(), writeOptions(false), - ); + )); await db.put_with_options( bytes("k2"), bytes("v2"), putOptions(), writeOptions(false), ); - const thirdWrite = await db.put_with_options( + const thirdWrite = cleanup.track(await db.put_with_options( bytes("k3"), bytes("v3"), putOptions(), writeOptions(false), - ); + )); await db.flush_with_options({ flush_type: FlushType.MemTable }); - const firstTimestamp = await admin.get_timestamp_for_sequence(firstWrite.seqnum, true); + const firstTimestamp = await admin.get_timestamp_for_sequence(firstWrite.seqnum(), true); assert.notEqual(firstTimestamp, undefined); const afterLastTimestamp = await admin.get_timestamp_for_sequence(MAX_U64, true); @@ -267,7 +267,7 @@ test("admin sequence lookups use persisted tracker", async (t) => { const seqAfterLast = await admin.get_sequence_for_timestamp(tomorrow, false); assert.notEqual(seqAfterLast, undefined); assert.ok(BigInt(seqAfterLast) > 0n); - assert.ok(BigInt(thirdWrite.seqnum) > BigInt(firstWrite.seqnum)); + assert.ok(BigInt(thirdWrite.seqnum()) > BigInt(firstWrite.seqnum())); const invalidTimestamp = await expectInvalid( () => admin.get_sequence_for_timestamp(MAX_I64, false), diff --git a/bindings/node/tests/db.test.mjs b/bindings/node/tests/db.test.mjs index d7c6b45a78..6e3d30b166 100644 --- a/bindings/node/tests/db.test.mjs +++ b/bindings/node/tests/db.test.mjs @@ -102,14 +102,15 @@ test("db crud and metadata", async (t) => { const store = cleanup.track(newMemoryStore()); const db = await openDb(store, { cleanup }); - const firstWrite = await db.put_with_options( + const firstWrite = cleanup.track(await db.put_with_options( bytes("alpha"), bytes("one"), putOptions(), writeOptions(false), - ); - assert.ok(firstWrite.seqnum > 0); - assert.ok(firstWrite.create_ts > 0); + )); + await firstWrite.await_durable(); + assert.ok(firstWrite.seqnum() > 0); + assert.ok(firstWrite.create_ts() > 0); assert.deepEqual(await db.get(bytes("alpha")), bytes("one")); assert.deepEqual( @@ -121,21 +122,21 @@ test("db crud and metadata", async (t) => { assert.notEqual(metadata, undefined); assert.deepEqual(metadata.key, bytes("alpha")); assert.deepEqual(metadata.value, bytes("one")); - assert.deepEqual(metadata.seq, firstWrite.seqnum); - assert.deepEqual(metadata.create_ts, firstWrite.create_ts); + assert.deepEqual(metadata.seq, firstWrite.seqnum()); + assert.deepEqual(metadata.create_ts, firstWrite.create_ts()); const metadataWithOptions = await db.get_key_value_with_options(bytes("alpha"), readOptions()); assert.notEqual(metadataWithOptions, undefined); assert.deepEqual(metadataWithOptions.value, bytes("one")); - const secondWrite = await db.put_with_options( + const secondWrite = cleanup.track(await db.put_with_options( bytes("beta"), bytes("two"), putOptions(), writeOptions(false), - ); - assert.ok(secondWrite.seqnum > firstWrite.seqnum); - assert.ok(secondWrite.create_ts > 0); + )); + assert.ok(secondWrite.seqnum() > firstWrite.seqnum()); + assert.ok(secondWrite.create_ts() > 0); assert.deepEqual(await db.get(bytes("beta")), bytes("two")); await db.put_with_options( @@ -147,12 +148,12 @@ test("db crud and metadata", async (t) => { assert.deepEqual(await db.get(bytes("empty")), bytes("")); assert.equal(await db.get(bytes("missing")), undefined); - const deleteAlpha = await db.delete_with_options(bytes("alpha"), writeOptions(false)); - assert.ok(deleteAlpha.seqnum > secondWrite.seqnum); + const deleteAlpha = cleanup.track(await db.delete_with_options(bytes("alpha"), writeOptions(false))); + assert.ok(deleteAlpha.seqnum() > secondWrite.seqnum()); assert.equal(await db.get(bytes("alpha")), undefined); - const deleteBeta = await db.delete_with_options(bytes("beta"), writeOptions(false)); - assert.ok(deleteBeta.seqnum > secondWrite.seqnum); + const deleteBeta = cleanup.track(await db.delete_with_options(bytes("beta"), writeOptions(false))); + assert.ok(deleteBeta.seqnum() > secondWrite.seqnum()); assert.equal(await db.get(bytes("beta")), undefined); }); @@ -249,8 +250,8 @@ test("db batch write and consumption", async (t) => { batch.put(bytes("batch-put"), bytes("value")); batch.delete(bytes("remove-me")); - const batchWrite = await db.write(batch); - assert.ok(batchWrite.seqnum > 0); + const batchWrite = cleanup.track(await db.write(batch)); + assert.ok(batchWrite.seqnum() > 0); assert.deepEqual(await db.get(bytes("batch-put")), bytes("value")); assert.equal(await db.get(bytes("remove-me")), undefined); @@ -261,7 +262,8 @@ test("db batch write and consumption", async (t) => { const secondBatch = cleanup.track(new WriteBatch()); secondBatch.put_with_options(bytes("batch-put-2"), bytes("value-2"), putOptions()); - await db.write_with_options(secondBatch, writeOptions()); + const secondBatchWrite = cleanup.track(await db.write_with_options(secondBatch, writeOptions())); + await secondBatchWrite.await_durable(); assert.deepEqual(await db.get(bytes("batch-put-2")), bytes("value-2")); }); @@ -301,12 +303,13 @@ test("db merge and merge_with_options", async (t) => { await db.merge(bytes("merge"), bytes(":one")); assert.deepEqual(await db.get(bytes("merge")), bytes("base:one")); - await db.merge_with_options( + const mergeWrite = cleanup.track(await db.merge_with_options( bytes("merge"), bytes(":two"), mergeOptions(), writeOptions(), - ); + )); + await mergeWrite.await_durable(); assert.deepEqual(await db.get(bytes("merge")), bytes("base:one:two")); }); @@ -346,9 +349,9 @@ test("db transactions", async (t) => { assert.deepEqual(await transaction.get(bytes("txn-key")), bytes("pending")); assert.equal(await db.get(bytes("txn-key")), undefined); - const commitHandle = await transaction.commit(); + const commitHandle = cleanup.track(await transaction.commit()); assert.notEqual(commitHandle, undefined); - assert.ok(commitHandle.seqnum > 0); + assert.ok(commitHandle.seqnum() > 0); assert.deepEqual(await db.get(bytes("txn-key")), bytes("pending")); const rollbackTx = cleanup.track(await db.begin(IsolationLevel.Snapshot)); @@ -423,7 +426,10 @@ test("db writer fencing reports closed reason", async (t) => { ); const error = await expectClosed( - () => primary.put(bytes("stale"), bytes("value")), + async () => { + const write = cleanup.track(await primary.put(bytes("stale"), bytes("value"))); + await write.await_durable(); + }, { reason: CloseReason.Fenced }, ); assert.match(error.message, /detected newer DB client/); diff --git a/bindings/node/tests/support.mjs b/bindings/node/tests/support.mjs index facc730c25..c90bde437d 100644 --- a/bindings/node/tests/support.mjs +++ b/bindings/node/tests/support.mjs @@ -69,9 +69,9 @@ export function readerOptions(skipWalReplay) { }; } -export function writeOptions(awaitDurable = true) { +export function writeOptions() { return { - await_durable: awaitDurable, + seqnum: 0, }; } diff --git a/bindings/python/tests/conftest.py b/bindings/python/tests/conftest.py index 9ddd245f34..5c3a49c038 100644 --- a/bindings/python/tests/conftest.py +++ b/bindings/python/tests/conftest.py @@ -77,7 +77,7 @@ def reader_options(skip_wal_replay: bool) -> ReaderOptions: def write_options() -> WriteOptions: - return WriteOptions(await_durable=True) + return WriteOptions() def put_options() -> PutOptions: diff --git a/bindings/python/tests/test_admin.py b/bindings/python/tests/test_admin.py index a2ab51df63..add2db69e0 100644 --- a/bindings/python/tests/test_admin.py +++ b/bindings/python/tests/test_admin.py @@ -73,7 +73,7 @@ async def test_admin_manifest_read_list_and_state_view() -> None: latest = await admin.read_manifest(None) assert latest is not None assert latest.id >= 3 - assert latest.last_l0_seq >= second_write.seqnum + assert latest.last_l0_seq >= second_write.seqnum() first = await admin.read_manifest(1) assert first is not None @@ -95,7 +95,7 @@ async def test_admin_manifest_read_list_and_state_view() -> None: state_view = await admin.read_compactor_state_view() assert state_view.manifest.id == latest.id - assert first_write.seqnum > 0 + assert first_write.seqnum() > 0 @pytest.mark.asyncio @@ -200,7 +200,7 @@ async def test_admin_sequence_lookups_use_persisted_tracker() -> None: third_write = await db.put(b"k3", b"v3") await db.flush_with_options(FlushOptions(flush_type=FlushType.MEM_TABLE)) - first_timestamp = await admin.get_timestamp_for_sequence(first_write.seqnum, True) + first_timestamp = await admin.get_timestamp_for_sequence(first_write.seqnum(), True) assert first_timestamp is not None after_last_timestamp = await admin.get_timestamp_for_sequence(MAX_U64, True) @@ -212,7 +212,7 @@ async def test_admin_sequence_lookups_use_persisted_tracker() -> None: seq_after_last = await admin.get_sequence_for_timestamp(int(time.time()) + 86_400, False) assert seq_after_last is not None assert seq_after_last > 0 - assert third_write.seqnum > first_write.seqnum + assert third_write.seqnum() > first_write.seqnum() with pytest.raises(Error.Invalid) as invalid_timestamp: await admin.get_sequence_for_timestamp(MAX_I64, False) diff --git a/bindings/python/tests/test_db.py b/bindings/python/tests/test_db.py index 8d969e123a..36f545aa91 100644 --- a/bindings/python/tests/test_db.py +++ b/bindings/python/tests/test_db.py @@ -96,8 +96,9 @@ async def test_db_crud_and_metadata() -> None: async with open_db(store) as db: first_write = await db.put(b"alpha", b"one") - assert first_write.seqnum > 0 - assert first_write.create_ts > 0 + await first_write.await_durable() + assert first_write.seqnum() > 0 + assert first_write.create_ts() > 0 assert await db.get(b"alpha") == b"one" assert await db.get_with_options(b"alpha", read_options()) == b"one" @@ -106,8 +107,8 @@ async def test_db_crud_and_metadata() -> None: assert metadata is not None assert metadata.key == b"alpha" assert metadata.value == b"one" - assert metadata.seq == first_write.seqnum - assert metadata.create_ts == first_write.create_ts + assert metadata.seq == first_write.seqnum() + assert metadata.create_ts == first_write.create_ts() metadata_with_options = await db.get_key_value_with_options(b"alpha", read_options()) assert metadata_with_options is not None @@ -119,8 +120,9 @@ async def test_db_crud_and_metadata() -> None: put_options(), write_options(), ) - assert second_write.seqnum > first_write.seqnum - assert second_write.create_ts > 0 + await second_write.await_durable() + assert second_write.seqnum() > first_write.seqnum() + assert second_write.create_ts() > 0 assert await db.get(b"beta") == b"two" await db.put(b"empty", b"") @@ -128,11 +130,12 @@ async def test_db_crud_and_metadata() -> None: assert await db.get(b"missing") is None delete_alpha = await db.delete(b"alpha") - assert delete_alpha.seqnum > second_write.seqnum + assert delete_alpha.seqnum() > second_write.seqnum() assert await db.get(b"alpha") is None delete_beta = await db.delete_with_options(b"beta", write_options()) - assert delete_beta.seqnum > second_write.seqnum + await delete_beta.await_durable() + assert delete_beta.seqnum() > second_write.seqnum() assert await db.get(b"beta") is None @@ -248,7 +251,7 @@ async def test_db_batch_write_and_consumption() -> None: batch.delete(b"remove-me") batch_write = await db.write(batch) - assert batch_write.seqnum > 0 + assert batch_write.seqnum() > 0 assert await db.get(b"batch-put") == b"value" assert await db.get(b"remove-me") is None @@ -258,7 +261,8 @@ async def test_db_batch_write_and_consumption() -> None: second_batch = WriteBatch() second_batch.put_with_options(b"batch-put-2", b"value-2", put_options()) - await db.write_with_options(second_batch, write_options()) + second_batch_write = await db.write_with_options(second_batch, write_options()) + await second_batch_write.await_durable() assert await db.get(b"batch-put-2") == b"value-2" @@ -285,12 +289,13 @@ async def test_db_merge_and_merge_with_options() -> None: await db.merge(b"merge", b":one") assert await db.get(b"merge") == b"base:one" - await db.merge_with_options( + merge_write = await db.merge_with_options( b"merge", b":two", merge_options(), write_options(), ) + await merge_write.await_durable() assert await db.get(b"merge") == b"base:one:two" @@ -322,7 +327,7 @@ async def test_db_transactions() -> None: commit_handle = await tx.commit() assert commit_handle is not None - assert commit_handle.seqnum > 0 + assert commit_handle.seqnum() > 0 assert await db.get(b"txn-key") == b"pending" rollback_tx = await db.begin(IsolationLevel.SNAPSHOT) @@ -385,7 +390,8 @@ async def test_db_writer_fencing_reports_closed_reason() -> None: await secondary.put(b"secondary", b"value") with pytest.raises(Error.Closed) as exc: - await primary.put(b"stale", b"value") + write = await primary.put(b"stale", b"value") + await write.await_durable() assert exc.value.reason == CloseReason.FENCED assert "detected newer DB client" in exc.value.message diff --git a/bindings/uniffi/src/config.rs b/bindings/uniffi/src/config.rs index 9abed81d5e..6b5ddcdf43 100644 --- a/bindings/uniffi/src/config.rs +++ b/bindings/uniffi/src/config.rs @@ -305,26 +305,18 @@ impl TryFrom for slatedb::config::ScanOptions { } } -/// Options that control durability behavior for writes and commits. -#[derive(Clone, Debug, uniffi::Record)] +/// Options that control writes and commits. +#[derive(Clone, Debug, Default, uniffi::Record)] pub struct WriteOptions { - /// Whether the call waits for the write to become durable before returning. - pub await_durable: bool, -} - -impl Default for WriteOptions { - fn default() -> Self { - Self { - await_durable: true, - } - } + /// Optional caller-supplied sequence number. Zero uses SlateDB's sequence oracle. + #[uniffi(default = 0)] + pub seqnum: u64, } impl From for slatedb::config::WriteOptions { fn from(value: WriteOptions) -> Self { slatedb::config::WriteOptions { - await_durable: value.await_durable, - ..Default::default() + seqnum: value.seqnum, } } } diff --git a/bindings/uniffi/src/db.rs b/bindings/uniffi/src/db.rs index c8b3b748f9..4e4dd22406 100644 --- a/bindings/uniffi/src/db.rs +++ b/bindings/uniffi/src/db.rs @@ -7,9 +7,10 @@ use crate::db_snapshot::DbSnapshot; use crate::db_transaction::DbTransaction; use crate::error::Error; use crate::iterator::DbIterator; -use crate::types::{CacheTarget, DbStatus, KeyRange, KeyValue, SsTableId, WriteHandle}; +use crate::types::{CacheTarget, DbStatus, KeyRange, KeyValue, SsTableId}; use crate::validation::{validate_key, validate_key_value}; use crate::write_batch::WriteBatch; +use crate::write_handle::WriteHandle; use slatedb::DbCacheManagerOps; /// A writable SlateDB handle. @@ -135,9 +136,11 @@ impl Db { /// /// Keys must be non-empty and at most `u16::MAX` bytes. Values must be at /// most `u32::MAX` bytes. - pub async fn put(&self, key: Vec, value: Vec) -> Result { + pub async fn put(&self, key: Vec, value: Vec) -> Result, Error> { validate_key_value(&key, &value)?; - Ok(self.inner.put(key, value).await?.into()) + Ok(Arc::new(WriteHandle::new( + self.inner.put(key, value).await?, + ))) } /// Inserts or overwrites a value using custom put and write options. @@ -147,21 +150,21 @@ impl Db { value: Vec, put_options: PutOptions, write_options: WriteOptions, - ) -> Result { + ) -> Result, Error> { validate_key_value(&key, &value)?; let put_options = put_options.into(); let write_options = write_options.into(); - Ok(self - .inner - .put_with_options(key, value, &put_options, &write_options) - .await? - .into()) + Ok(Arc::new(WriteHandle::new( + self.inner + .put_with_options(key, value, &put_options, &write_options) + .await?, + ))) } /// Deletes `key` and returns metadata for the write. - pub async fn delete(&self, key: Vec) -> Result { + pub async fn delete(&self, key: Vec) -> Result, Error> { validate_key(&key)?; - Ok(self.inner.delete(key).await?.into()) + Ok(Arc::new(WriteHandle::new(self.inner.delete(key).await?))) } /// Deletes `key` using custom write options. @@ -169,16 +172,20 @@ impl Db { &self, key: Vec, options: WriteOptions, - ) -> Result { + ) -> Result, Error> { validate_key(&key)?; let options = options.into(); - Ok(self.inner.delete_with_options(key, &options).await?.into()) + Ok(Arc::new(WriteHandle::new( + self.inner.delete_with_options(key, &options).await?, + ))) } /// Appends a merge operand for `key` and returns metadata for the write. - pub async fn merge(&self, key: Vec, operand: Vec) -> Result { + pub async fn merge(&self, key: Vec, operand: Vec) -> Result, Error> { validate_key_value(&key, &operand)?; - Ok(self.inner.merge(key, operand).await?.into()) + Ok(Arc::new(WriteHandle::new( + self.inner.merge(key, operand).await?, + ))) } /// Appends a merge operand using custom merge and write options. @@ -188,23 +195,23 @@ impl Db { operand: Vec, merge_options: MergeOptions, write_options: WriteOptions, - ) -> Result { + ) -> Result, Error> { validate_key_value(&key, &operand)?; let merge_options = merge_options.into(); let write_options = write_options.into(); - Ok(self - .inner - .merge_with_options(key, operand, &merge_options, &write_options) - .await? - .into()) + Ok(Arc::new(WriteHandle::new( + self.inner + .merge_with_options(key, operand, &merge_options, &write_options) + .await?, + ))) } /// Applies all operations in `batch` atomically. /// /// The provided batch is consumed and cannot be reused afterwards. - pub async fn write(&self, batch: Arc) -> Result { + pub async fn write(&self, batch: Arc) -> Result, Error> { let batch = batch.take_for_write()?; - Ok(self.inner.write(batch).await?.into()) + Ok(Arc::new(WriteHandle::new(self.inner.write(batch).await?))) } /// Applies all operations in `batch` atomically using custom write options. @@ -214,10 +221,12 @@ impl Db { &self, batch: Arc, options: WriteOptions, - ) -> Result { + ) -> Result, Error> { let batch = batch.take_for_write()?; let options = options.into(); - Ok(self.inner.write_with_options(batch, &options).await?.into()) + Ok(Arc::new(WriteHandle::new( + self.inner.write_with_options(batch, &options).await?, + ))) } /// Flushes the default storage layer. diff --git a/bindings/uniffi/src/db_transaction.rs b/bindings/uniffi/src/db_transaction.rs index 338ca115ba..4dad26aec8 100644 --- a/bindings/uniffi/src/db_transaction.rs +++ b/bindings/uniffi/src/db_transaction.rs @@ -5,8 +5,9 @@ use tokio::sync::Mutex; use crate::config::{MergeOptions, PutOptions, ReadOptions, ScanOptions, WriteOptions}; use crate::error::{Error, SlateDbError}; use crate::iterator::DbIterator; -use crate::types::{KeyRange, KeyValue, WriteHandle}; +use crate::types::{KeyRange, KeyValue}; use crate::validation::{validate_key, validate_key_value}; +use crate::write_handle::WriteHandle; /// Transaction handle returned by [`crate::Db::begin`]. /// @@ -226,12 +227,15 @@ impl DbTransaction { /// Commits the transaction. /// /// Returns `None` when the transaction performed no writes. - pub async fn commit(&self) -> Result, Error> { + pub async fn commit(&self) -> Result>, Error> { let tx = { let mut guard = self.inner.lock().await; guard.take().ok_or(SlateDbError::TransactionCompleted)? }; - Ok(tx.commit().await?.map(WriteHandle::from)) + Ok(tx + .commit() + .await? + .map(|handle| Arc::new(WriteHandle::new(handle)))) } /// Commits the transaction using custom write options. @@ -240,7 +244,7 @@ impl DbTransaction { pub async fn commit_with_options( &self, options: WriteOptions, - ) -> Result, Error> { + ) -> Result>, Error> { let options = options.into(); let tx = { let mut guard = self.inner.lock().await; @@ -249,6 +253,6 @@ impl DbTransaction { Ok(tx .commit_with_options(&options) .await? - .map(WriteHandle::from)) + .map(|handle| Arc::new(WriteHandle::new(handle)))) } } diff --git a/bindings/uniffi/src/lib.rs b/bindings/uniffi/src/lib.rs index 310e4bc947..bf28ac7a4b 100644 --- a/bindings/uniffi/src/lib.rs +++ b/bindings/uniffi/src/lib.rs @@ -19,6 +19,7 @@ mod types; mod validation; mod wal_reader; mod write_batch; +mod write_handle; pub use admin::Admin; pub use builder::{AdminBuilder, CloneBuilder, DbBuilder, DbReaderBuilder}; @@ -50,9 +51,10 @@ pub use types::{ CompactorStateView, CompressionCodec, DbStatus, ExternalDb, FilterFormat, IdentifiedObjectMetadata, KeyRange, KeyValue, ObjectMetadata, RowEntry, RowEntryKind, Segment, SegmentPrefix, SortedRun, SourceId, SsTableHandle, SsTableId, SsTableInfo, SsTableView, - SstType, VersionedCompactions, VersionedManifest, WriteHandle, + SstType, VersionedCompactions, VersionedManifest, }; pub use wal_reader::{WalFile, WalFileIterator, WalReader}; pub use write_batch::WriteBatch; +pub use write_handle::WriteHandle; uniffi::setup_scaffolding!("slatedb"); diff --git a/bindings/uniffi/src/types.rs b/bindings/uniffi/src/types.rs index 4a33f42be7..640f59e2b6 100644 --- a/bindings/uniffi/src/types.rs +++ b/bindings/uniffi/src/types.rs @@ -108,24 +108,6 @@ impl KeyRange { } } -/// Metadata returned by a successful write. -#[derive(Clone, Debug, PartialEq, Eq, uniffi::Record)] -pub struct WriteHandle { - /// Sequence number assigned to the write. - pub seqnum: u64, - /// Creation timestamp assigned to the write. - pub create_ts: i64, -} - -impl From for WriteHandle { - fn from(value: slatedb::WriteHandle) -> Self { - Self { - seqnum: value.seqnum(), - create_ts: value.create_ts(), - } - } -} - /// A segment (RFC-0024), identified by the key prefix it owns; the segment /// spans the key interval `[prefix, prefix++)`. #[derive(Clone, Debug, PartialEq, Eq, uniffi::Record)] diff --git a/bindings/uniffi/src/write_handle.rs b/bindings/uniffi/src/write_handle.rs new file mode 100644 index 0000000000..cc8d1cdf6d --- /dev/null +++ b/bindings/uniffi/src/write_handle.rs @@ -0,0 +1,34 @@ +use crate::error::Error; + +/// Handle returned by a successful write. +#[derive(uniffi::Object)] +pub struct WriteHandle { + inner: slatedb::WriteHandle, +} + +impl WriteHandle { + pub(crate) fn new(inner: slatedb::WriteHandle) -> Self { + Self { inner } + } +} + +#[uniffi::export] +impl WriteHandle { + /// Returns the sequence number assigned to the write. + pub fn seqnum(&self) -> u64 { + self.inner.seqnum() + } + + /// Returns the creation timestamp assigned to the write. + pub fn create_ts(&self) -> i64 { + self.inner.create_ts() + } +} + +#[uniffi::export(async_runtime = "tokio")] +impl WriteHandle { + /// Waits until the write has been durably persisted. + pub async fn await_durable(&self) -> Result<(), Error> { + self.inner.await_durable().await.map_err(Into::into) + } +} diff --git a/examples/src/azure_blob_storage.rs b/examples/src/azure_blob_storage.rs index 4efc2604d6..135db85d53 100644 --- a/examples/src/azure_blob_storage.rs +++ b/examples/src/azure_blob_storage.rs @@ -25,7 +25,6 @@ async fn main() -> anyhow::Result<()> { // Put 1000 keys, do not wait for it to be durable println!("Writing 1000 keys without waiting for flush"); let write_options = slatedb::config::WriteOptions { - await_durable: false, ..Default::default() }; for i in 0..1000 { diff --git a/examples/src/google_cloud_storage.rs b/examples/src/google_cloud_storage.rs index 11ede8629e..f24fe87bd9 100644 --- a/examples/src/google_cloud_storage.rs +++ b/examples/src/google_cloud_storage.rs @@ -24,7 +24,6 @@ async fn main() -> anyhow::Result<()> { // Put 1000 keys, do not wait for it to be durable println!("Writing 1000 keys without waiting for flush"); let write_options = slatedb::config::WriteOptions { - await_durable: false, ..Default::default() }; for i in 0..1000 { diff --git a/slatedb-bencher/src/db.rs b/slatedb-bencher/src/db.rs index 9b521bc074..bacda18249 100644 --- a/slatedb-bencher/src/db.rs +++ b/slatedb-bencher/src/db.rs @@ -162,6 +162,7 @@ pub struct DbBench { key_gen_supplier: Box Box>, val_len: usize, write_options: WriteOptions, + await_durable: bool, concurrency: u32, num_rows: Option, duration: Option, @@ -176,6 +177,7 @@ impl DbBench { key_gen_supplier: Box Box>, val_len: usize, write_options: WriteOptions, + await_durable: bool, concurrency: u32, num_rows: Option, duration: Option, @@ -187,6 +189,7 @@ impl DbBench { key_gen_supplier, val_len, write_options, + await_durable, concurrency, num_rows, duration, @@ -209,6 +212,7 @@ impl DbBench { (*self.key_gen_supplier)(), self.val_len, self.write_options.clone(), + self.await_durable, self.num_rows, self.duration, self.put_percentage, @@ -229,6 +233,7 @@ struct Task { key_generator: Box, val_len: usize, write_options: WriteOptions, + await_durable: bool, num_keys: Option, duration: Option, put_percentage: u32, @@ -243,6 +248,7 @@ impl Task { key_generator: Box, val_len: usize, write_options: WriteOptions, + await_durable: bool, num_keys: Option, duration: Option, put_percentage: u32, @@ -254,6 +260,7 @@ impl Task { key_generator, val_len, write_options, + await_durable, num_keys, duration, put_percentage, @@ -283,12 +290,17 @@ impl Task { let key = self.key_generator.next_key(); let mut value = vec![0; self.val_len]; random.fill_bytes(value.as_mut_slice()); - match self + let result = self .db .put_with_options(key, value, &PutOptions::default(), &self.write_options) - .await - { - Ok(_) => { + .await; + let result = match result { + Ok(handle) if self.await_durable => handle.await_durable().await, + Ok(_) => Ok(()), + Err(error) => Err(error), + }; + match result { + Ok(()) => { puts += 1; puts_bytes += self.val_len as u64; } diff --git a/slatedb-bencher/src/main.rs b/slatedb-bencher/src/main.rs index 5101648cc4..22d085ad9b 100644 --- a/slatedb-bencher/src/main.rs +++ b/slatedb-bencher/src/main.rs @@ -83,10 +83,7 @@ async fn exec_benchmark_db(path: Path, object_store: Arc, args: if args.no_compactor { config.compactor_options = None; } - let write_options = WriteOptions { - await_durable: args.await_durable, - ..Default::default() - }; + let write_options = WriteOptions::default(); let mut builder = Db::builder(path.clone(), object_store.clone()).with_settings(config); @@ -99,6 +96,7 @@ async fn exec_benchmark_db(path: Path, object_store: Arc, args: args.key_gen_supplier(), args.val_len, write_options, + args.await_durable, args.concurrency, args.num_rows, args.duration.map(|d| Duration::from_secs(d as u64)), @@ -156,10 +154,7 @@ async fn exec_benchmark_transaction( args: BenchmarkTransactionArgs, ) { let (config, memory_cache) = args.db_args.config().unwrap(); - let write_options = WriteOptions { - await_durable: args.await_durable, - ..Default::default() - }; + let write_options = WriteOptions::default(); let mut builder = Db::builder(path.clone(), object_store.clone()).with_settings(config); @@ -173,6 +168,7 @@ async fn exec_benchmark_transaction( args.key_gen_supplier(), args.val_len, write_options, + args.await_durable, args.concurrency, args.duration.map(|d| Duration::from_secs(d as u64)), args.transaction_size, diff --git a/slatedb-bencher/src/transactions.rs b/slatedb-bencher/src/transactions.rs index 97fa0b7b78..9b621ca565 100644 --- a/slatedb-bencher/src/transactions.rs +++ b/slatedb-bencher/src/transactions.rs @@ -48,6 +48,7 @@ pub struct TransactionBench { key_gen_supplier: Box Box>, val_len: usize, write_options: WriteOptions, + await_durable: bool, concurrency: u32, duration: Option, transaction_size: u32, @@ -63,6 +64,7 @@ impl TransactionBench { key_gen_supplier: Box Box>, val_len: usize, write_options: WriteOptions, + await_durable: bool, concurrency: u32, duration: Option, transaction_size: u32, @@ -75,6 +77,7 @@ impl TransactionBench { key_gen_supplier, val_len, write_options, + await_durable, concurrency, duration, transaction_size, @@ -98,6 +101,7 @@ impl TransactionBench { (*self.key_gen_supplier)(), self.val_len, self.write_options.clone(), + self.await_durable, self.duration, self.transaction_size, self.abort_percentage, @@ -119,6 +123,7 @@ struct TransactionTask { key_generator: Box, val_len: usize, write_options: WriteOptions, + await_durable: bool, duration: Option, transaction_size: u32, abort_percentage: u32, @@ -134,6 +139,7 @@ impl TransactionTask { key_generator: Box, val_len: usize, write_options: WriteOptions, + await_durable: bool, duration: Option, transaction_size: u32, abort_percentage: u32, @@ -146,6 +152,7 @@ impl TransactionTask { key_generator, val_len, write_options, + await_durable, duration, transaction_size, abort_percentage, @@ -231,8 +238,14 @@ impl TransactionTask { batch.put(key, value); } - match self.db.write_with_options(batch, &self.write_options).await { - Ok(_) => Ok(ops as u64), + let result = self.db.write_with_options(batch, &self.write_options).await; + let result = match result { + Ok(handle) if self.await_durable => handle.await_durable().await, + Ok(_) => Ok(()), + Err(error) => Err(error), + }; + match result { + Ok(()) => Ok(ops as u64), Err(e) => { warn!("write batch failed [error={}]", e); Err(e) @@ -271,8 +284,14 @@ impl TransactionTask { return TransactionResult::Aborted; } - match txn.commit_with_options(&self.write_options).await { - Ok(_) => TransactionResult::Committed(ops as u64), + let result = txn.commit_with_options(&self.write_options).await; + let result = match result { + Ok(Some(handle)) if self.await_durable => handle.await_durable().await, + Ok(_) => Ok(()), + Err(error) => Err(error), + }; + match result { + Ok(()) => TransactionResult::Committed(ops as u64), Err(e) => { warn!("transaction commit failed (conflict) [error={}]", e); TransactionResult::Conflict diff --git a/slatedb-dst/src/actors/bank/mod.rs b/slatedb-dst/src/actors/bank/mod.rs index 5e76650089..df303b1dec 100644 --- a/slatedb-dst/src/actors/bank/mod.rs +++ b/slatedb-dst/src/actors/bank/mod.rs @@ -86,7 +86,6 @@ pub async fn initialize_accounts(db: &Db, options: &BankOptions) -> Result<(), E &starting_balance, &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) diff --git a/slatedb-dst/src/actors/bank/transfer.rs b/slatedb-dst/src/actors/bank/transfer.rs index 39515be947..5c82b2d402 100644 --- a/slatedb-dst/src/actors/bank/transfer.rs +++ b/slatedb-dst/src/actors/bank/transfer.rs @@ -40,7 +40,6 @@ impl TransferActor { impl Actor for TransferActor { async fn run(&mut self, ctx: &ActorCtx) -> Result<(), Error> { let write_options = WriteOptions { - await_durable: false, ..WriteOptions::default() }; let from_rand = ctx.rand().rng().next_u64(); diff --git a/slatedb-dst/src/actors/fencer.rs b/slatedb-dst/src/actors/fencer.rs index 5b3be8b596..c7aeaa3837 100644 --- a/slatedb-dst/src/actors/fencer.rs +++ b/slatedb-dst/src/actors/fencer.rs @@ -80,7 +80,11 @@ impl Actor for DbFencerActor { let old_db = ctx.swap_db(next_db); // Verify the old DB is fenced. - match old_db.put(b"foo", b"bar").await { + let result = match old_db.put(b"foo", b"bar").await { + Ok(handle) => handle.await_durable().await, + Err(error) => Err(error), + }; + match result { Err(err) if matches!(err.kind(), ErrorKind::Closed(CloseReason::Fenced)) => {} result => panic!("old db was not fenced as expected [result={result:?}]"), } diff --git a/slatedb-dst/src/actors/workload.rs b/slatedb-dst/src/actors/workload.rs index a4221d59d7..6634717210 100644 --- a/slatedb-dst/src/actors/workload.rs +++ b/slatedb-dst/src/actors/workload.rs @@ -162,7 +162,6 @@ impl Actor for WorkloadActor { async fn run(&mut self, ctx: &ActorCtx) -> Result<(), Error> { let put_options = PutOptions::default(); let write_options = WriteOptions { - await_durable: false, ..WriteOptions::default() }; let key_prefix = self diff --git a/slatedb/benches/db_operations.rs b/slatedb/benches/db_operations.rs index 6f1055ccba..e37c4ec4e2 100644 --- a/slatedb/benches/db_operations.rs +++ b/slatedb/benches/db_operations.rs @@ -25,7 +25,6 @@ fn criterion_benchmark(c: &mut Criterion) { value, &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) diff --git a/slatedb/benches/db_transaction.rs b/slatedb/benches/db_transaction.rs index c0230ebcce..edb3f33f99 100644 --- a/slatedb/benches/db_transaction.rs +++ b/slatedb/benches/db_transaction.rs @@ -65,7 +65,6 @@ fn merge_options() -> MergeOptions { fn write_options() -> WriteOptions { WriteOptions { - await_durable: false, ..WriteOptions::default() } } diff --git a/slatedb/benches/scan_prefix_bench.rs b/slatedb/benches/scan_prefix_bench.rs index f5b391a4da..a85bbd7b86 100644 --- a/slatedb/benches/scan_prefix_bench.rs +++ b/slatedb/benches/scan_prefix_bench.rs @@ -136,10 +136,7 @@ fn recency_scan_options() -> ScanOptions { } async fn populate(db: &Db) { - let write_opts = WriteOptions { - await_durable: false, - seqnum: 0, - }; + let write_opts = WriteOptions::default(); let put_opts = PutOptions::default(); let mut next_version = [0u64; NUM_PREFIXES]; for round in 0..NUM_FLUSHES { diff --git a/slatedb/src/admin.rs b/slatedb/src/admin.rs index 7555ba7895..8ba01bc1d7 100644 --- a/slatedb/src/admin.rs +++ b/slatedb/src/admin.rs @@ -1408,7 +1408,6 @@ mod tests { ..Settings::default() }; let write_opts = WriteOptions { - await_durable: false, ..Default::default() }; diff --git a/slatedb/src/batch_write.rs b/slatedb/src/batch_write.rs index 00a067fd13..651ed6d3b3 100644 --- a/slatedb/src/batch_write.rs +++ b/slatedb/src/batch_write.rs @@ -29,9 +29,7 @@ use async_trait::async_trait; use fail_parallel::fail_point; use futures::stream::BoxStream; use futures::{FutureExt, StreamExt}; -use log::warn; use std::sync::Arc; -use std::time::Duration; use tracing::instrument; use std::collections::BTreeSet; @@ -47,7 +45,6 @@ use crate::wal::{FlushResultFuture, WalWriter}; use crate::{batch::WriteBatch, db::DbInner, db::WriteHandle, error::SlateDBError}; use bytes::Bytes; use parking_lot::RwLockWriteGuard; -use slatedb_common::clock::SystemClock; use tokio::sync::oneshot; pub(crate) const WRITE_BATCH_TASK_NAME: &str = "writer"; @@ -102,7 +99,6 @@ impl std::fmt::Debug for BatchWriterMessage { pub(crate) struct WriteBatchEventHandler { db_inner: Arc, - is_first_write: bool, wal_writer: Option>, } @@ -110,7 +106,6 @@ impl WriteBatchEventHandler { pub(crate) fn new(db_inner: Arc, wal_writer: Option>) -> Self { Self { db_inner, - is_first_write: true, wal_writer, } } @@ -128,15 +123,8 @@ impl MessageHandler for WriteBatchEventHandler { }) => { let result = self .db_inner - .write_batch( - batch, - &options, - txn.as_ref(), - self.wal_writer.as_mut(), - self.is_first_write, - ) + .write_batch(batch, &options, txn.as_ref(), self.wal_writer.as_mut()) .await; - self.is_first_write = false; match result { Ok(write_result) => { let _ = done.send(write_result); @@ -207,7 +195,6 @@ impl DbInner { options: &WriteOptions, txn: Option<&DbTransaction>, wal_writer: Option<&mut Box>, - is_first_write: bool, ) -> Result { let _options = options; #[cfg(not(dst))] @@ -278,17 +265,7 @@ impl DbInner { self.write_entries_to_memtable(entries, touched_segments); } else { assert!(!self.wal_enabled); - // if WAL is disabled, we just write the entries to memtable. - let watcher = self.write_entries_to_memtable(entries, touched_segments); - // if this is the first write and the WAL is disabled, make sure users are flushing - // their memtables in a timely manner. - if is_first_write && options.await_durable { - let this_watcher = watcher.clone(); - let this_clock = self.system_clock.clone(); - tokio::spawn(async move { - monitor_first_write(this_watcher, this_clock).await; - }); - } + self.write_entries_to_memtable(entries, touched_segments); }; // increment memtable_write_bytes by the size of the keys and values inserted into the memtable // after merge operators and overwrites are collapsed @@ -328,7 +305,8 @@ impl DbInner { // maybe freeze the memtable. self.maybe_freeze_current_memtable()?; - let write_handle = WriteHandle::new(commit_seq, now); + let write_handle = + WriteHandle::new_with_waiter(commit_seq, now, self.status_manager.durability_waiter()); Ok(Ok(write_handle)) } @@ -521,21 +499,6 @@ fn check_segment_prefix_antichain( Ok(()) } -async fn monitor_first_write( - mut watcher: WatchableOnceCellReader>, - system_clock: Arc, -) { - tokio::select! { - _ = watcher.await_value() => {} - _ = system_clock.sleep(Duration::from_secs(5)) => { - warn!("First write not durable after 5 seconds and WAL is disabled. \ - SlateDB does not automatically flush memtables until `l0_sst_size_bytes` \ - is reached. If writer is single threaded or has low throughput, the \ - applications must call `flush` to ensure durability in a timely manner."); - } - } -} - #[cfg(test)] mod tests { use super::*; @@ -613,31 +576,6 @@ mod tests { ) } - #[tokio::test] - async fn test_is_first_write_set_false_after_first_write() { - let object_store = Arc::new(InMemory::new()); - let db = Db::open( - "/tmp/test_is_first_write_set_false_after_first_write", - object_store, - ) - .await - .unwrap(); - - let wal_writer = Box::new(FakeWalWriter::new(0)); - let mut handler = WriteBatchEventHandler::new(db.inner.clone(), Some(wal_writer)); - assert!(handler.is_first_write); - - let mut batch = WriteBatch::new(); - batch.put(b"key", b"value"); - - let (msg, done_rx) = test_message(batch, WriteOptions::default()); - handler.handle(msg).await.unwrap(); - - let result = done_rx.await.unwrap(); - assert!(result.is_ok()); - assert!(!handler.is_first_write); - } - #[tokio::test] async fn test_append_error_notifies_caller_and_fails_handler() { let object_store = Arc::new(InMemory::new()); @@ -709,8 +647,9 @@ mod tests { let (msg, done_rx) = test_message( batch, WriteOptions { + #[cfg(dst)] + now: 0, seqnum: 42, - ..Default::default() }, ); handler.handle(msg).await.unwrap(); @@ -753,8 +692,9 @@ mod tests { let (msg, done_rx) = test_message( batch, WriteOptions { + #[cfg(dst)] + now: 0, seqnum: 1, - ..Default::default() }, ); handler.handle(msg).await.unwrap(); diff --git a/slatedb/src/clone.rs b/slatedb/src/clone.rs index c37ce2d613..d9d65041a5 100644 --- a/slatedb/src/clone.rs +++ b/slatedb/src/clone.rs @@ -1197,7 +1197,6 @@ mod tests { .await .unwrap(); let write_options = WriteOptions { - await_durable: false, ..Default::default() }; let put_options = PutOptions::default(); @@ -1288,7 +1287,6 @@ mod tests { .await .unwrap(); let write_options = WriteOptions { - await_durable: false, ..Default::default() }; let put_options = PutOptions::default(); @@ -1385,10 +1383,7 @@ mod tests { .build() .await .unwrap(); - let write_options = WriteOptions { - await_durable: false, - ..Default::default() - }; + let write_options = WriteOptions::default(); let put_options = PutOptions::default(); // Keys inside and outside the projection range [aaa, bbb), flushed @@ -1559,7 +1554,7 @@ mod tests { .build() .await .unwrap(); - // await_durable would deadlock under wal_enabled=false because the + // Do not await the returned handle here: with wal_enabled=false, the // memtable flush is gated on the explicit call below. test_utils::seed_database(&db, table, false).await.unwrap(); if wal_enabled { diff --git a/slatedb/src/compactor.rs b/slatedb/src/compactor.rs index 0611510f78..2312ce90ac 100644 --- a/slatedb/src/compactor.rs +++ b/slatedb/src/compactor.rs @@ -1672,7 +1672,6 @@ mod tests { &v, &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -1685,7 +1684,6 @@ mod tests { &v, &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -1813,15 +1811,9 @@ mod tests { for key in [b"a", b"b", b"c", b"d"] { batch.put(key, b"value"); } - db.write_with_options( - batch, - &WriteOptions { - await_durable: false, - ..Default::default() - }, - ) - .await - .unwrap(); + db.write_with_options(batch, &WriteOptions::default()) + .await + .unwrap(); db.flush_with_options(FlushOptions { flush_type: FlushType::MemTable, }) @@ -2312,8 +2304,8 @@ mod tests { let (manifest_store, _, table_store) = build_test_stores(os.clone()); - // put key 'a' into L1 (and key 'b' so that when we delete 'a' the SST is non-empty) - // since these are both await_durable=true, we're guaranteed to have one L0 SST for each. + // Put key 'a' into L1 (and key 'b' so that when we delete 'a' the SST is non-empty). + // The explicit flush below makes both writes durable. db.put(&[b'a'; 16], &[b'a'; 32]).await.unwrap(); db.put(&[b'b'; 16], &[b'a'; 32]).await.unwrap(); db.flush().await.unwrap(); @@ -2327,7 +2319,6 @@ mod tests { db.delete_with_options( &[b'a'; 16], &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -2441,7 +2432,6 @@ mod tests { db.delete_with_options( &[b'a'; 16], &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -2534,7 +2524,6 @@ mod tests { b"a", &MergeOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -2545,7 +2534,6 @@ mod tests { b"b", &MergeOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -2556,7 +2544,6 @@ mod tests { &vec![b'p'; 128], &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -2569,7 +2556,6 @@ mod tests { b"c", &MergeOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -2580,7 +2566,6 @@ mod tests { b"x", &MergeOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -2591,7 +2576,6 @@ mod tests { &vec![b'p'; 128], &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -2676,7 +2660,6 @@ mod tests { b"a", &MergeOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -2693,7 +2676,6 @@ mod tests { b"b", &MergeOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -2746,7 +2728,6 @@ mod tests { b"a", &MergeOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -2757,7 +2738,6 @@ mod tests { b"b", &MergeOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -2768,7 +2748,6 @@ mod tests { &vec![b'p'; 128], &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -2789,7 +2768,6 @@ mod tests { b"c", &MergeOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -2800,7 +2778,6 @@ mod tests { b"d", &MergeOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -2811,7 +2788,6 @@ mod tests { &vec![b'p'; 128], &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -2887,7 +2863,6 @@ mod tests { b"x", &MergeOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -2898,7 +2873,6 @@ mod tests { &vec![b'p'; 128], &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -2911,7 +2885,6 @@ mod tests { b"y", &MergeOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -2922,7 +2895,6 @@ mod tests { b"z", &MergeOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -2933,7 +2905,6 @@ mod tests { &vec![b'p'; 128], &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -3009,7 +2980,6 @@ mod tests { b"1", &MergeOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -3020,7 +2990,6 @@ mod tests { &vec![b'p'; 128], &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -3033,7 +3002,6 @@ mod tests { b"2", &MergeOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -3044,7 +3012,6 @@ mod tests { &vec![b'p'; 128], &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -3057,7 +3024,6 @@ mod tests { b"3", &MergeOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -3068,7 +3034,6 @@ mod tests { &vec![b'p'; 128], &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -3154,11 +3119,13 @@ mod tests { ttl: Ttl::ExpireAfterMillis(10), }, &WriteOptions { - await_durable: true, ..Default::default() }, ) .await + .unwrap() + .await_durable() + .await .unwrap(); // ticker time = 20, no expire time @@ -3168,11 +3135,13 @@ mod tests { &[b'b'; 32], &crate::config::MergeOptions { ttl: Ttl::NoExpiry }, &WriteOptions { - await_durable: true, ..Default::default() }, ) .await + .unwrap() + .await_durable() + .await .unwrap(); let db_state = await_compaction(&db, os.clone(), Some(insert_clock.clone())) @@ -3237,7 +3206,6 @@ mod tests { b"a", &MergeOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -3248,7 +3216,6 @@ mod tests { b"b", &MergeOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -3259,7 +3226,6 @@ mod tests { &vec![b'p'; 128], &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -3273,7 +3239,6 @@ mod tests { b"new_value", &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -3284,7 +3249,6 @@ mod tests { &vec![b'p'; 128], &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -3346,7 +3310,6 @@ mod tests { ttl: Ttl::ExpireAfterMillis(100), }, &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -3357,7 +3320,6 @@ mod tests { &vec![b'p'; 128], &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -3372,7 +3334,6 @@ mod tests { ttl: Ttl::ExpireAfterMillis(200), }, &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -3383,7 +3344,6 @@ mod tests { &vec![b'p'; 128], &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -3487,7 +3447,6 @@ mod tests { ttl: Ttl::ExpireAtMillis(1000), }, &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -3503,7 +3462,6 @@ mod tests { ttl: Ttl::ExpireAtMillis(1000), }, &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -3604,7 +3562,6 @@ mod tests { ttl: Ttl::ExpireAtMillis(10), }, &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -3620,7 +3577,6 @@ mod tests { ttl: Ttl::ExpireAtMillis(i64::MAX), }, &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -3636,7 +3592,6 @@ mod tests { value, &PutOptions { ttl: Ttl::NoExpiry }, &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -3720,7 +3675,6 @@ mod tests { ttl: Ttl::ExpireAfterMillis(10), }, &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -3734,7 +3688,6 @@ mod tests { value, &PutOptions { ttl: Ttl::Default }, &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -3750,7 +3703,6 @@ mod tests { value, &PutOptions { ttl: Ttl::NoExpiry }, &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -3767,7 +3719,6 @@ mod tests { ttl: Ttl::ExpireAfterMillis(80), }, &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -6165,7 +6116,6 @@ mod tests { value, &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) diff --git a/slatedb/src/config.rs b/slatedb/src/config.rs index 483245c839..83bd501651 100644 --- a/slatedb/src/config.rs +++ b/slatedb/src/config.rs @@ -463,11 +463,8 @@ impl Default for FlushOptions { /// Configuration for client write operations. `WriteOptions` is supplied for each /// write call and controls the behavior of the write. -#[derive(Clone, Debug)] +#[derive(Clone, Debug, Default)] pub struct WriteOptions { - /// Whether `put` calls should block until the write has been durably committed - /// to the DB. - pub await_durable: bool, #[cfg(dst)] /// Force the current timestamp for DST operations. See #719 for details. pub now: i64, @@ -478,18 +475,6 @@ pub struct WriteOptions { pub seqnum: u64, } -impl Default for WriteOptions { - /// Create a new `WriteOptions`` with `await_durable` set to `true`. - fn default() -> Self { - Self { - await_durable: true, - #[cfg(dst)] - now: 0, - seqnum: 0, - } - } -} - /// Configuration for client put operations. `PutOptions` is supplied for each /// row inserted. This differs from [`WriteOptions`] in that a write may encompass /// multiple puts (such as the case with batched writes) diff --git a/slatedb/src/db.rs b/slatedb/src/db.rs index d6782ef363..e638eb0916 100644 --- a/slatedb/src/db.rs +++ b/slatedb/src/db.rs @@ -23,6 +23,8 @@ pub use crate::db_status::{DbStatus, SegmentPrefix}; use crate::db_cache::CacheTarget; +use std::fmt; +use std::future::Future; use std::sync::Arc; use bytes::Bytes; @@ -75,7 +77,7 @@ use slatedb_common::metrics::MetricsRecorderHelper; use slatedb_common::DbRand; use slatedb_txn_obj::DirtyObject; -use crate::db_status::{ClosedResultWriter, DbStatusManager}; +use crate::db_status::{ClosedResultWriter, DbStatusManager, DurabilityWaiter}; use crate::wal::{WalEvent, WalIterator, WalObserver, WalStatus}; pub use builder::DbBuilder; pub use builder::DbReaderBuilder; @@ -297,28 +299,7 @@ impl DbInner { self.maybe_apply_backpressure().await?; self.write_notifier.send(batch_msg)?; - // TODO: this can be modified as awaiting the last_durable_seq watermark & fatal error. - - let write_handle = rx.await??; - - if options.await_durable { - let seq = write_handle.seq; - let mut status_subscription = self.status_manager.subscribe(); - let status = status_subscription - .wait_for(|s| s.durable_seq >= seq || s.close_reason.is_some()) - .await - .map_err(|_| SlateDBError::Closed)?; - if status.durable_seq < seq { - self.check_closed()?; - warn!( - "durable seq {} not advanced past write seq {} and db not closed", - status.durable_seq, seq - ); - return Err(SlateDBError::InvalidDBState); - } - } - - Ok(write_handle) + rx.await? } #[inline] @@ -1341,7 +1322,13 @@ impl Db { .map_err(Into::into) } - /// Write a value into the database with default `WriteOptions`. + /// Write a value into the database with default `PutOptions` and + /// `WriteOptions`. + /// + /// This method returns after updating the in-memory WAL and MemTable. It + /// does not wait for the write to become durable in object storage. Call + /// [`WriteHandle::await_durable`] on the returned handle to wait for this + /// write, or [`Db::flush`] to flush all pending writes. /// /// ## Arguments /// - `key`: the key to write @@ -1377,6 +1364,11 @@ impl Db { /// Write a value into the database with custom `PutOptions` and `WriteOptions`. /// + /// This method returns after updating the in-memory WAL and MemTable. It + /// does not wait for the write to become durable in object storage. Call + /// [`WriteHandle::await_durable`] on the returned handle to wait for this + /// write, or [`Db::flush`] to flush all pending writes. + /// /// ## Arguments /// - `key`: the key to write /// - `value`: the value to write @@ -1422,6 +1414,10 @@ impl Db { /// this form when the caller already holds the data as [`Bytes`] (e.g. /// from a prior read, a zero-copy buffer pool, or a client that produces /// [`Bytes`] directly). + /// + /// This method does not wait for durability. Call + /// [`WriteHandle::await_durable`] on the returned handle to wait for this + /// write, or [`Db::flush`] to flush all pending writes. pub async fn put_bytes(&self, key: Bytes, value: Bytes) -> Result { self.put_bytes_with_options(key, value, &PutOptions::default(), &WriteOptions::default()) .await @@ -1430,6 +1426,10 @@ impl Db { /// Write a value into the database using owned [`Bytes`] with custom /// `PutOptions` and `WriteOptions`. See [`Db::put_bytes`] for why this /// form exists. + /// + /// This method does not wait for durability. Call + /// [`WriteHandle::await_durable`] on the returned handle to wait for this + /// write, or [`Db::flush`] to flush all pending writes. pub async fn put_bytes_with_options( &self, key: Bytes, @@ -1444,6 +1444,11 @@ impl Db { /// Delete a key from the database with default `WriteOptions`. /// + /// This method returns after updating the in-memory WAL and MemTable. It + /// does not wait for the delete to become durable in object storage. Call + /// [`WriteHandle::await_durable`] on the returned handle to wait for this + /// delete, or [`Db::flush`] to flush all pending writes. + /// /// ## Arguments /// - `key`: the key to delete /// @@ -1473,6 +1478,11 @@ impl Db { /// Delete a key from the database with custom `WriteOptions`. /// + /// This method returns after updating the in-memory WAL and MemTable. It + /// does not wait for the delete to become durable in object storage. Call + /// [`WriteHandle::await_durable`] on the returned handle to wait for this + /// delete, or [`Db::flush`] to flush all pending writes. + /// /// ## Arguments /// - `key`: the key to delete /// - `options`: the write options to use @@ -1507,6 +1517,11 @@ impl Db { /// Merge a value into the database with default `MergeOptions` and `WriteOptions`. /// + /// This method returns after updating the in-memory WAL and MemTable. It + /// does not wait for the merge to become durable in object storage. Call + /// [`WriteHandle::await_durable`] on the returned handle to wait for this + /// merge, or [`Db::flush`] to flush all pending writes. + /// /// Merge operations allow applications to bypass the traditional read/modify/write cycle /// by expressing partial updates using an associative operator. The merge operator must /// be configured when opening the database. @@ -1563,6 +1578,11 @@ impl Db { /// Merge a value into the database with custom `MergeOptions` and `WriteOptions`. /// + /// This method returns after updating the in-memory WAL and MemTable. It + /// does not wait for the merge to become durable in object storage. Call + /// [`WriteHandle::await_durable`] on the returned handle to wait for this + /// merge, or [`Db::flush`] to flush all pending writes. + /// /// Merge operations allow applications to bypass the traditional read/modify/write cycle /// by expressing partial updates using an associative operator. The merge operator must /// be configured when opening the database. @@ -1634,6 +1654,11 @@ impl Db { /// block other gets and writes until the batch is written to the WAL (or memtable if /// WAL is disabled). /// + /// This method returns after updating the in-memory WAL and MemTable. It + /// does not wait for the batch to become durable in object storage. Call + /// [`WriteHandle::await_durable`] on the returned handle to wait for this + /// batch, or [`Db::flush`] to flush all pending writes. + /// /// ## Arguments /// - `batch`: the batch of put/delete operations to write /// @@ -1670,6 +1695,11 @@ impl Db { /// block other gets and writes until the batch is written to the WAL (or memtable if /// WAL is disabled). /// + /// This method returns after updating the in-memory WAL and MemTable. It + /// does not wait for the batch to become durable in object storage. Call + /// [`WriteHandle::await_durable`] on the returned handle to wait for this + /// batch, or [`Db::flush`] to flush all pending writes. + /// /// ## Arguments /// - `batch`: the batch of put/delete operations to write /// - `options`: the write options to use @@ -1709,8 +1739,8 @@ impl Db { .map_err(Into::into) } - /// Flush in-memory writes to disk. This function blocks until the in-memory - /// data has been durably written to object storage. + /// Flush in-memory writes to object storage. This function blocks until + /// the in-memory data has been durably written. /// /// If WAL is enabled, this method is equivalent to: /// `flush_with_options(FlushOptions { flush_type: FlushType::Wal })` @@ -2030,16 +2060,58 @@ impl DbCacheManagerOps for Db { } /// Handle returned from write operations, containing metadata about the write. +/// +/// Write operations return this handle without waiting for durability. Call +/// [`WriteHandle::await_durable`] to wait until this write is durable in object +/// storage. +/// /// This structure is designed to be extensible for future enhancements. -#[derive(Debug, Clone)] +#[derive(Clone)] pub struct WriteHandle { pub(crate) seq: u64, pub(crate) create_ts: i64, + durability_waiter: DurabilityWaiter, +} + +impl fmt::Debug for WriteHandle { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.debug_struct("WriteHandle") + .field("seq", &self.seq) + .field("create_ts", &self.create_ts) + .finish_non_exhaustive() + } } impl WriteHandle { - pub fn new(seq: u64, create_ts: i64) -> Self { - Self { seq, create_ts } + /// Creates a write handle that uses `durability_waiter` to determine when + /// the write is durable. + /// + /// This constructor is intended for custom [`DbWriteOps`] implementations + /// and test doubles whose writes are not immediately durable. + pub fn new(seq: u64, create_ts: i64, durability_waiter: F) -> Self + where + F: Fn() -> Fut + Send + Sync + 'static, + Fut: Future> + Send + 'static, + { + let durability_waiter: DurabilityWaiter = Arc::new(move |_| Box::pin(durability_waiter())); + + Self { + seq, + create_ts, + durability_waiter, + } + } + + pub(crate) fn new_with_waiter( + seq: u64, + create_ts: i64, + durability_waiter: DurabilityWaiter, + ) -> Self { + Self { + seq, + create_ts, + durability_waiter, + } } /// Returns the sequence number assigned to this write operation. @@ -2051,6 +2123,18 @@ impl WriteHandle { pub fn create_ts(&self) -> i64 { self.create_ts } + + /// Waits until this write has been durably persisted. + /// + /// If the database closes before the write becomes durable, this returns a + /// closed error carrying the database's [`CloseReason`]. + /// + /// # Errors + /// + /// Returns a closed error if the database closes first. + pub async fn await_durable(&self) -> Result<(), crate::Error> { + (self.durability_waiter)(self.seq).await + } } /// Wraps [`WalObserver`] and injects a [`crate::wal_buffer::WalStatusListener`] @@ -2860,10 +2944,7 @@ mod tests { b"px:b", b"vb_dirty", &PutOptions::default(), - &WriteOptions { - await_durable: false, - seqnum: 0, - }, + &WriteOptions::default(), ) .await .unwrap(); @@ -3040,7 +3121,6 @@ mod tests { &value, &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -3090,7 +3170,6 @@ mod tests { value, &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -3143,7 +3222,6 @@ mod tests { b"test_value", &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -3227,7 +3305,6 @@ mod tests { b"test_value", &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -3293,7 +3370,6 @@ mod tests { b"test_value", &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -3351,7 +3427,6 @@ mod tests { b"test_value", &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -3486,7 +3561,6 @@ mod tests { b"world", &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -3534,7 +3608,6 @@ mod tests { .unwrap(); let put_options = PutOptions::default(); let write_options = WriteOptions { - await_durable: false, ..Default::default() }; let get_memory_options = ReadOptions::new().with_durability_filter(Memory); @@ -3594,7 +3667,6 @@ mod tests { ttl: Default::default(), }, &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -3690,7 +3762,7 @@ mod tests { async fn build_database_from_table( table: &BTreeMap, db_options: Settings, - await_durable: bool, + wait_for_durability: bool, ) -> Db { let object_store: Arc = Arc::new(InMemory::new()); let db = Db::builder("/tmp/test_kv_store", object_store) @@ -3701,7 +3773,7 @@ mod tests { test_utils::seed_database(&db, table, false).await.unwrap(); - if await_durable { + if wait_for_durability { db.flush().await.unwrap(); } @@ -4132,7 +4204,6 @@ mod tests { db.delete_with_options( &[b'b'; 4], &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -4184,7 +4255,6 @@ mod tests { .await .unwrap(); let write_options = WriteOptions { - await_durable: false, ..Default::default() }; @@ -4204,7 +4274,6 @@ mod tests { // the memtable will not be flushed to l0, and the test will hang // at this put_with_options call. let write_options = WriteOptions { - await_durable: true, ..Default::default() }; clock.set(10); @@ -4215,6 +4284,9 @@ mod tests { &write_options, ) .await + .unwrap() + .await_durable() + .await .unwrap(); let state = wait_for_manifest_condition( @@ -4280,10 +4352,22 @@ mod tests { for i in 0..3 { let key = [b'a' + i; 16]; let value = [b'b' + i; 50]; - kv_store.put(&key, &value).await.unwrap(); + kv_store + .put(&key, &value) + .await + .unwrap() + .await_durable() + .await + .unwrap(); let key = [b'j' + i; 16]; let value = [b'k' + i; 50]; - kv_store.put(&key, &value).await.unwrap(); + kv_store + .put(&key, &value) + .await + .unwrap() + .await_durable() + .await + .unwrap(); let db_state = wait_for_manifest_condition( &mut stored_manifest, |s| s.replay_after_wal_id > last_wal_id, @@ -4348,7 +4432,6 @@ mod tests { .unwrap(); let write_options: WriteOptions = WriteOptions { - await_durable: false, ..Default::default() }; let put_options = PutOptions::default(); @@ -4477,7 +4560,6 @@ mod tests { value1, &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -4492,7 +4574,6 @@ mod tests { value2, &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -4594,7 +4675,6 @@ mod tests { value, &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -4679,7 +4759,6 @@ mod tests { .await .unwrap(); let write_options = WriteOptions { - await_durable: false, ..Default::default() }; let put_options = PutOptions::default(); @@ -4792,7 +4871,6 @@ mod tests { .unwrap(); let write_options = WriteOptions { - await_durable: false, ..Default::default() }; @@ -4923,7 +5001,6 @@ mod tests { value1, &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -4938,7 +5015,6 @@ mod tests { value2, &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -5012,7 +5088,6 @@ mod tests { value1, &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -5083,7 +5158,6 @@ mod tests { .unwrap(); let metrics_recorder_clone = metrics_recorder.clone(); let write_opts = WriteOptions { - await_durable: false, ..Default::default() }; @@ -5205,7 +5279,6 @@ mod tests { .await .unwrap(); let write_opts = WriteOptions { - await_durable: false, ..Default::default() }; @@ -5359,7 +5432,6 @@ mod tests { // do all flushes manually let write_opts = WriteOptions { - await_durable: false, ..Default::default() }; @@ -5467,7 +5539,6 @@ mod tests { b"val1", &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -5478,7 +5549,6 @@ mod tests { b"val2", &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -5489,7 +5559,6 @@ mod tests { b"val3", &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -5623,7 +5692,6 @@ mod tests { "bar".as_bytes(), &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -5669,6 +5737,9 @@ mod tests { kv_store .put("foo".as_bytes(), "bar".as_bytes()) .await + .unwrap() + .await_durable() + .await .unwrap(); fail_parallel::cfg(fp_registry.clone(), "write-wal-sst-io-error", "pause").unwrap(); kv_store @@ -5677,7 +5748,6 @@ mod tests { "bla".as_bytes(), &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -5720,13 +5790,15 @@ mod tests { kv_store .put("foo".as_bytes(), "bar".as_bytes()) .await + .unwrap() + .await_durable() + .await .unwrap(); fail_parallel::cfg(fp_registry.clone(), "write-wal-sst-io-error", "pause").unwrap(); kv_store .delete_with_options( "foo".as_bytes(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -5778,6 +5850,9 @@ mod tests { kv_store .put("key3".as_bytes(), "committed3".as_bytes()) .await + .unwrap() + .await_durable() + .await .unwrap(); // Pause WAL writes to prevent new writes from being committed @@ -5790,7 +5865,6 @@ mod tests { "uncommitted2".as_bytes(), &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -5802,7 +5876,6 @@ mod tests { "uncommitted4".as_bytes(), &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -5886,11 +5959,21 @@ mod tests { // write a few keys that will result in memtable flushes let key1 = [b'a'; 32]; let value1 = [b'b'; 96]; - db.put(key1, value1).await.unwrap(); + db.put(key1, value1) + .await + .unwrap() + .await_durable() + .await + .unwrap(); next_wal_id += 1; let key2 = [b'c'; 32]; let value2 = [b'd'; 96]; - db.put(key2, value2).await.unwrap(); + db.put(key2, value2) + .await + .unwrap() + .await_durable() + .await + .unwrap(); next_wal_id += 1; let reader = Db::builder(path, object_store.clone()) @@ -5974,6 +6057,7 @@ mod tests { let value1 = [b'b'; 96]; let result = db.put(&key1, &value1).await; assert!(result.is_ok(), "Failed to write key1"); + result.unwrap().await_durable().await.unwrap(); assert_eq!( db.inner.wal_observer.status().unwrap().last_flushed_wal_id, 2 @@ -6043,7 +6127,13 @@ mod tests { ); fail_parallel::cfg(fp_registry.clone(), "write-wal-sst-io-error", "panic").unwrap(); - let result = db.put(b"foo", b"bar").await.unwrap_err(); + let result = db + .put(b"foo", b"bar") + .await + .unwrap() + .await_durable() + .await + .unwrap_err(); assert!(result.to_string().contains("background task panicked")); } @@ -6061,12 +6151,23 @@ mod tests { .unwrap(), ); // Trigger a WAL write and block until durable so WAL is written - db.put(b"foo", b"bar").await.unwrap(); + db.put(b"foo", b"bar") + .await + .unwrap() + .await_durable() + .await + .unwrap(); fail_parallel::cfg(fp_registry.clone(), "write-wal-sst-io-error", "panic").unwrap(); // Trigger a WAL write, which should not advance the manifest WAL ID - let result = db.put(b"foo", b"bar").await.unwrap_err(); + let result = db + .put(b"foo", b"bar") + .await + .unwrap() + .await_durable() + .await + .unwrap_err(); assert_eq!(result.kind(), crate::ErrorKind::Closed(CloseReason::Panic)); assert!(result .to_string() @@ -6123,7 +6224,6 @@ mod tests { b"bar", &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -6136,15 +6236,56 @@ mod tests { .expect_err("close should error out due to WAL IO error"); } + #[tokio::test] + async fn test_constructed_write_handle_uses_durability_waiter() { + let waiter_called = Arc::new(AtomicBool::new(false)); + let waiter_called_clone = waiter_called.clone(); + let handle = WriteHandle::new(1, 0, move || { + let waiter_called = waiter_called_clone.clone(); + async move { + waiter_called.store(true, Ordering::SeqCst); + Ok(()) + } + }); + + handle.await_durable().await.unwrap(); + + assert!(waiter_called.load(Ordering::SeqCst)); + } + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] - async fn test_await_durable_write_returns_error_if_db_closes_before_durable() { + async fn test_write_handle_await_durable_waits_for_flush() { + let object_store: Arc = Arc::new(InMemory::new()); + let mut settings = test_db_options(0, 1024, None); + settings.flush_interval = None; + let db = Db::builder( + "/tmp/test_write_handle_await_durable_waits_for_flush", + object_store, + ) + .with_settings(settings) + .build() + .await + .unwrap(); + + let handle = db.put(b"foo", b"bar").await.unwrap(); + let durability_wait = tokio::spawn(async move { handle.await_durable().await }); + tokio::task::yield_now().await; + assert!(!durability_wait.is_finished()); + + db.flush().await.unwrap(); + durability_wait.await.unwrap().unwrap(); + db.close().await.unwrap(); + } + + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] + async fn test_write_handle_await_durable_returns_error_if_db_closes_first() { let fp_registry = Arc::new(FailPointRegistry::new()); let object_store: Arc = Arc::new(InMemory::new()); let mut settings = test_db_options(0, 1024, None); settings.flush_interval = None; let db = Arc::new( Db::builder( - "/tmp/test_await_durable_write_returns_error_if_db_closes_before_durable", + "/tmp/test_write_handle_await_durable_returns_error_if_db_closes_first", object_store, ) .with_settings(settings) @@ -6158,14 +6299,15 @@ mod tests { fail_parallel::cfg(fp_registry.clone(), "write-wal-sst-io-error", "pause").unwrap(); let write_db = db.clone(); let write_task = tokio::spawn(async move { - write_db + let handle = write_db .put_with_options( b"foo", b"bar", &PutOptions::default(), &WriteOptions::default(), ) - .await + .await?; + handle.await_durable().await }); tokio::time::timeout(Duration::from_secs(10), async { loop { @@ -6237,8 +6379,7 @@ mod tests { |s| { // compact after writing values. include in loop since the on demand scheduler // only runs once per `should_compact`, and memtables might still be getting - // flushed (await_durable in the put()'s above only wait for the writes to hit - // the WAL before returning). + // flushed (the put() calls above return before the writes become durable). should_compact_l0.store(true, Ordering::SeqCst); s.tree.last_compacted_l0_sst_view_id.is_some() && s.tree.l0.is_empty() }, @@ -6263,8 +6404,7 @@ mod tests { |s| { // compact after writing values. include in loop since the on demand scheduler // only runs once per `should_compact`, and memtables might still be getting - // flushed (await_durable in the put()'s above only wait for the writes to hit - // the WAL before returning). + // flushed (the put() calls above return before the writes become durable). should_compact_l0.store(true, Ordering::SeqCst); s.tree.last_compacted_l0_sst_view_id.is_some() && s.tree.l0.is_empty() }, @@ -6377,16 +6517,11 @@ mod tests { let path = "/tmp/test_kv_store"; async fn do_put(db: &Db, key: &[u8], val: &[u8]) -> Result { - db.put_with_options( - key, - val, - &PutOptions::default(), - &WriteOptions { - await_durable: true, - ..Default::default() - }, - ) - .await + let handle = db + .put_with_options(key, val, &PutOptions::default(), &WriteOptions::default()) + .await?; + handle.await_durable().await?; + Ok(handle) } // open db1 and assert that it can write. @@ -6503,13 +6638,12 @@ mod tests { b"w1", b"value", &PutOptions::default(), - &WriteOptions { - await_durable: true, - ..Default::default() - }, + &WriteOptions::default(), ) .await; - assert!(result.is_err()); + if let Ok(handle) = result { + assert!(handle.await_durable().await.is_err()); + } } async fn wait_for_wal_sst_count(table_store: &TableStore, min_count: usize, context: &str) { @@ -6687,7 +6821,6 @@ mod tests { b"1", &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -6703,7 +6836,7 @@ mod tests { b"1", &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() + ..Default::default() }, ) .await @@ -6735,7 +6868,6 @@ mod tests { b"1", &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -6756,7 +6888,7 @@ mod tests { b"1", &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() + ..Default::default() }, ) .await @@ -6837,7 +6969,6 @@ mod tests { &[b'j'; 8], &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -6849,7 +6980,6 @@ mod tests { &[b'k'; 8], &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -7078,14 +7208,6 @@ mod tests { } } - #[test] - fn test_write_option_defaults() { - // This is a regression test for a bug where the defaults for WriteOptions were not being - // set correctly due to visibility issues. - let write_options = WriteOptions::default(); - assert!(write_options.await_durable); - } - #[tokio::test] #[cfg(feature = "zstd")] async fn test_compression_overflow_bug() { @@ -7110,7 +7232,6 @@ mod tests { let value = format!("{}{}", "v".repeat(i), i); let put_option = PutOptions::default(); let write_option = WriteOptions { - await_durable: false, ..Default::default() }; db.put_with_options(key.as_bytes(), value.clone(), &put_option, &write_option) @@ -7713,7 +7834,6 @@ mod tests { // do a write and flush memtable only (not wal) let write_opts = WriteOptions { - await_durable: false, ..Default::default() }; db.put_with_options(&b"foo", &b"bar", &PutOptions::default(), &write_opts) @@ -7944,7 +8064,6 @@ mod tests { b"value1", &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -8114,7 +8233,6 @@ mod tests { value, &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -8134,7 +8252,6 @@ mod tests { b"value2", &put_opts, &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -8149,7 +8266,6 @@ mod tests { .delete_with_options( b"key1", &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -8167,7 +8283,6 @@ mod tests { .write_with_options( batch, &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -8198,7 +8313,6 @@ mod tests { .write_with_options( batch, &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -8216,7 +8330,6 @@ mod tests { .write_with_options( batch, &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -8233,7 +8346,6 @@ mod tests { .write_with_options( batch, &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -8257,7 +8369,6 @@ mod tests { .write_with_options( WriteBatch::new(), &WriteOptions { - await_durable: false, ..Default::default() }, None, @@ -8634,7 +8745,6 @@ mod tests { .await .unwrap(); let write_opts = WriteOptions { - await_durable: false, ..Default::default() }; @@ -9036,7 +9146,6 @@ mod tests { b"value1", &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -9047,7 +9156,6 @@ mod tests { b"value2", &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -9137,7 +9245,6 @@ mod tests { b"aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa0", &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -9153,7 +9260,6 @@ mod tests { val.as_bytes(), &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -9293,7 +9399,6 @@ mod tests { b"base", &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -9311,7 +9416,6 @@ mod tests { operand, &MergeOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -9411,7 +9515,6 @@ mod tests { ttl: Ttl::ExpireAfterMillis(50), }, &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -9444,7 +9547,6 @@ mod tests { ttl: Ttl::ExpireAfterMillis(50), }; let write_opts = WriteOptions { - await_durable: false, ..Default::default() }; @@ -9520,7 +9622,6 @@ mod tests { ttl: Ttl::ExpireAtMillis(500), }, &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -9535,7 +9636,6 @@ mod tests { ttl: Ttl::ExpireAtMillis(500), }, &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -9660,7 +9760,6 @@ mod tests { db.write_with_options( batch, &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -9746,7 +9845,6 @@ mod tests { // when: two writes (the second triggers maybe_apply_backpressure for the first's bytes) let write_opts = WriteOptions { - await_durable: false, ..Default::default() }; db.put_with_options(b"k1", b"v1", &PutOptions::default(), &write_opts) @@ -9787,7 +9885,6 @@ mod tests { b"v1", &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -9799,7 +9896,6 @@ mod tests { b"v2", &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -9839,7 +9935,6 @@ mod tests { b"v1", &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -10005,7 +10100,6 @@ mod tests { .unwrap(); let write_opts = WriteOptions { - await_durable: false, ..Default::default() }; @@ -10165,7 +10259,6 @@ mod tests { &value, &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -10204,7 +10297,6 @@ mod tests { b"value", &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -10232,7 +10324,6 @@ mod tests { .await .unwrap(); let write_opts = WriteOptions { - await_durable: false, ..Default::default() }; for (key, value) in [ @@ -10304,7 +10395,6 @@ mod tests { .await .unwrap(); let write_opts = WriteOptions { - await_durable: false, ..Default::default() }; source @@ -10353,7 +10443,6 @@ mod tests { let mut rx = db.subscribe(); assert!(rx.borrow_and_update().list_segments().is_empty()); let write_opts = WriteOptions { - await_durable: false, ..Default::default() }; @@ -10410,7 +10499,6 @@ mod tests { b"v1", &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -10478,7 +10566,6 @@ mod tests { b"v1", &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -10788,7 +10875,6 @@ mod tests { .await .unwrap(); let write_opts = WriteOptions { - await_durable: false, ..Default::default() }; for (k, v) in [(b"aaa-1".as_slice(), b"v1"), (b"bbb-1".as_slice(), b"v2")] { diff --git a/slatedb/src/db_cache_manager.rs b/slatedb/src/db_cache_manager.rs index e9929ba629..773274b4e1 100644 --- a/slatedb/src/db_cache_manager.rs +++ b/slatedb/src/db_cache_manager.rs @@ -270,7 +270,6 @@ mod tests { &value, &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) diff --git a/slatedb/src/db_reader.rs b/slatedb/src/db_reader.rs index 734dee4387..f7119e1ce6 100644 --- a/slatedb/src/db_reader.rs +++ b/slatedb/src/db_reader.rs @@ -1424,7 +1424,12 @@ mod tests { let key = b"test_key"; let value = b"test_value"; - db.put(key, value).await.unwrap(); + db.put(key, value) + .await + .unwrap() + .await_durable() + .await + .unwrap(); db.flush().await.unwrap(); let reader = DbReader::open( @@ -1551,7 +1556,6 @@ mod tests { .await .unwrap(); let write_opts = WriteOptions { - await_durable: false, ..Default::default() }; db.put_with_options(b"abc-1", b"v1", &PutOptions::default(), &write_opts) @@ -1645,7 +1649,6 @@ mod tests { .await .unwrap(); let write_opts = WriteOptions { - await_durable: false, ..Default::default() }; db.put_with_options(b"abc-1", b"v1", &PutOptions::default(), &write_opts) @@ -2205,7 +2208,12 @@ mod tests { .unwrap(); let key = b"test_key"; let value = b"test_value"; - db.put(key, value).await.unwrap(); + db.put(key, value) + .await + .unwrap() + .await_durable() + .await + .unwrap(); db.flush().await.unwrap(); tokio::time::sleep(Duration::from_millis(500)).await; @@ -3299,7 +3307,6 @@ mod tests { b"a", &MergeOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -3310,7 +3317,6 @@ mod tests { b"b", &MergeOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -3353,7 +3359,6 @@ mod tests { b"c", &MergeOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -3364,7 +3369,6 @@ mod tests { b"d", &MergeOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) diff --git a/slatedb/src/db_snapshot.rs b/slatedb/src/db_snapshot.rs index 90834c5a4a..102f005d74 100644 --- a/slatedb/src/db_snapshot.rs +++ b/slatedb/src/db_snapshot.rs @@ -769,7 +769,6 @@ mod tests { b"value2", &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) diff --git a/slatedb/src/db_status.rs b/slatedb/src/db_status.rs index 2448db8ce6..dfc521bce2 100644 --- a/slatedb/src/db_status.rs +++ b/slatedb/src/db_status.rs @@ -1,11 +1,14 @@ use std::collections::BTreeSet; +use std::fmt; +use std::sync::Arc; use bytes::Bytes; +use futures::future::BoxFuture; use tokio::sync::watch; use crate::error::SlateDBError; use crate::manifest::VersionedManifest; -use crate::utils::WatchableOnceCell; +use crate::utils::{WatchableOnceCell, WatchableOnceCellReader}; use crate::CloseReason; /// A segment (RFC-0024), identified by the key prefix it owns; the segment @@ -71,10 +74,20 @@ pub(crate) trait ClosedResultWriter: std::fmt::Debug + Send + Sync + 'static { /// Manages database lifecycle status, including the close result and /// status subscriptions. -#[derive(Clone, Debug)] +#[derive(Clone)] pub(crate) struct DbStatusManager { cell: WatchableOnceCell>, tx: watch::Sender, + durability_waiter: DurabilityWaiter, +} + +impl fmt::Debug for DbStatusManager { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.debug_struct("DbStatusManager") + .field("cell", &self.cell) + .field("tx", &self.tx) + .finish_non_exhaustive() + } } impl DbStatusManager { @@ -103,9 +116,13 @@ impl DbStatusManager { memtable_segments: initial_memtable_segments, close_reason: None, }); + let cell = WatchableOnceCell::new(); + let durability_waiter = new_durability_waiter(tx.subscribe(), cell.reader()); + Self { - cell: WatchableOnceCell::new(), + cell, tx, + durability_waiter, } } @@ -214,6 +231,11 @@ impl DbStatusManager { self.tx.subscribe() } + /// Returns the shared [`DurabilityWaiter`] for this database. + pub(crate) fn durability_waiter(&self) -> DurabilityWaiter { + Arc::clone(&self.durability_waiter) + } + pub(crate) fn status(&self) -> DbStatus { self.tx.borrow().clone() } @@ -245,6 +267,45 @@ impl ClosedResultWriter for DbStatusManager { } } +/// Shared callback used by [`crate::WriteHandle`]s to wait for a sequence +/// number to become durable. +pub(crate) type DurabilityWaiter = + Arc BoxFuture<'static, Result<(), crate::Error>> + Send + Sync + 'static>; + +/// Creates a [`DurabilityWaiter`] backed by database status and close-result +/// readers. +/// +/// The waiter captures only readers, so cloning it into a write handle does not +/// keep the database's status sender alive after the database is dropped. +fn new_durability_waiter( + status_rx: watch::Receiver, + close_result: WatchableOnceCellReader>, +) -> DurabilityWaiter { + Arc::new(move |seq| -> BoxFuture<'static, Result<(), crate::Error>> { + let mut status_rx = status_rx.clone(); + let close_result = close_result.clone(); + + Box::pin(async move { + let wait_result = status_rx + .wait_for(|status| status.durable_seq >= seq || status.close_reason.is_some()) + .await; + + match wait_result { + Ok(status) if status.durable_seq >= seq => Ok(()), + // The write was not durable before the database closed. Use the + // recorded close result to preserve fencing and panic errors. + Ok(_) | Err(_) => match close_result + .read() + .expect("database closed without recording a close result") + { + Ok(()) => Err(SlateDBError::Closed.into()), + Err(error) => Err(error.into()), + }, + } + }) + }) +} + #[cfg(test)] mod tests { use super::*; diff --git a/slatedb/src/db_transaction.rs b/slatedb/src/db_transaction.rs index 0c25f2bb3f..5d9c1fdf02 100644 --- a/slatedb/src/db_transaction.rs +++ b/slatedb/src/db_transaction.rs @@ -21,6 +21,11 @@ use crate::{DbReadOps, DbTransactionOps}; /// configurable isolation levels. This is the main interface for transactional /// operations in SlateDB. /// +/// Committing a non-empty transaction returns a [`WriteHandle`] without +/// waiting for durability. Call [`WriteHandle::await_durable`] on that handle +/// to wait for that commit, or call [`crate::Db::flush`] to flush all pending +/// writes. +/// /// # Examples /// /// Basic transaction usage: @@ -516,10 +521,16 @@ impl DbTransaction { /// Commit the transaction by applying all buffered operations to the database. /// - /// This method finalizes the transaction by writing all pending puts, deletes, and other - /// operations from the write batch to persistent storage. The actual conflict detection - /// (including read-write and write-write conflicts) is deferred to the task that processes - /// the WriteBatch, which ensures the atomicity of transactions. + /// This method finalizes the transaction by writing all pending puts, + /// deletes, and other operations from the write batch to the in-memory WAL + /// and MemTable. The actual conflict detection (including read-write and + /// write-write conflicts) is deferred to the task that processes the + /// WriteBatch, which ensures the atomicity of transactions. + /// + /// A successful commit does not wait for durability in object storage. + /// Call [`WriteHandle::await_durable`] on the returned handle when the + /// result is `Some`, or call [`crate::Db::flush`] to flush all pending + /// writes. /// /// If the transaction's write batch is empty, this operation is a no-op and returns `Ok(())` /// immediately without any database interaction. Since it's impossible to have read-write @@ -540,7 +551,12 @@ impl DbTransaction { /// Commit the transaction with custom write options. /// /// This method behaves the same as [`DbTransaction::commit`], but allows callers - /// to specify custom [`WriteOptions`], such as `await_durable`. + /// to specify custom [`WriteOptions`]. + /// + /// A successful commit does not wait for durability in object storage. + /// Call [`WriteHandle::await_durable`] on the returned handle when the + /// result is `Some`, or call [`crate::Db::flush`] to flush all pending + /// writes. /// /// ## Arguments /// - `options`: the write options to use for the commit @@ -1174,14 +1190,14 @@ mod tests { } #[tokio::test(flavor = "multi_thread", worker_threads = 4)] - async fn test_txn_commit_await_durable_false() { + async fn test_txn_commit_returns_before_durable() { use crate::config::{DurabilityLevel::*, ReadOptions, WriteOptions}; use fail_parallel::FailPointRegistry; // Setup database with failpoints to pause durable writes let fp_registry = Arc::new(FailPointRegistry::new()); let object_store: Arc = Arc::new(InMemory::new()); - let db = crate::Db::builder("/tmp/test_txn_commit_await_durable_false", object_store) + let db = crate::Db::builder("/tmp/test_txn_commit_returns_before_durable", object_store) .with_fp_registry(fp_registry.clone()) .build() .await @@ -1194,13 +1210,10 @@ mod tests { let txn = db.begin(IsolationLevel::Snapshot).await.unwrap(); txn.put(b"k", b"v").unwrap(); - // Commit without waiting for durability - txn.commit_with_options(&WriteOptions { - await_durable: false, - ..Default::default() - }) - .await - .unwrap(); + // Commits return without waiting for durability. + txn.commit_with_options(&WriteOptions::default()) + .await + .unwrap(); // Memory (in-memory) read should see the value let val = db @@ -2124,7 +2137,6 @@ mod tests { txn.put(b"key1", b"value1").unwrap(); let handle = txn .commit_with_options(&WriteOptions { - await_durable: false, ..Default::default() }) .await @@ -2142,7 +2154,6 @@ mod tests { txn.put_with_options(b"key2", b"value2", &put_opts).unwrap(); let handle = txn .commit_with_options(&WriteOptions { - await_durable: false, ..Default::default() }) .await @@ -2157,7 +2168,6 @@ mod tests { txn.delete(b"key1").unwrap(); let handle = txn .commit_with_options(&WriteOptions { - await_durable: false, ..Default::default() }) .await @@ -2177,7 +2187,6 @@ mod tests { let txn = db.begin(IsolationLevel::Snapshot).await.unwrap(); let result = txn .commit_with_options(&WriteOptions { - await_durable: false, ..Default::default() }) .await diff --git a/slatedb/src/fence.rs b/slatedb/src/fence.rs index 80e4754664..399a982ae2 100644 --- a/slatedb/src/fence.rs +++ b/slatedb/src/fence.rs @@ -306,7 +306,10 @@ mod tests { async fn put(&mut self, db: &Db, v: u32, expect_fenced: bool) { let k = Bytes::from(format!("k{}", v)); let v = Bytes::from(format!("v{}", v)); - let result = db.put(k.as_ref(), v.as_ref()).await; + let result = match db.put(k.as_ref(), v.as_ref()).await { + Ok(handle) => handle.await_durable().await, + Err(error) => Err(error), + }; if expect_fenced { assert_eq!( result.unwrap_err().kind(), @@ -593,7 +596,10 @@ mod tests { // verify that fenced db is fenced (new write fails) use crate::error::{CloseReason, ErrorKind}; - let err = db.put(b"k4", b"v4").await.unwrap_err(); + let err = match db.put(b"k4", b"v4").await { + Ok(handle) => handle.await_durable().await.unwrap_err(), + Err(error) => error, + }; assert!( matches!(err.kind(), ErrorKind::Closed(CloseReason::Fenced)), "expected Fenced, got {err}" diff --git a/slatedb/src/ops.rs b/slatedb/src/ops.rs index dcdd1cd402..2e242e09e9 100644 --- a/slatedb/src/ops.rs +++ b/slatedb/src/ops.rs @@ -210,6 +210,11 @@ pub trait DbReadOps { /// This trait defines the asynchronous write API exposed by [`Db`](crate::Db), /// allowing consumers to write generic code or test doubles over the writer /// surface without depending on the concrete `Db` type. +/// +/// Write methods return after updating the in-memory WAL and MemTable. They do +/// not wait for the write to become durable in object storage. Call +/// [`WriteHandle::await_durable`] on the returned handle to wait for one write, +/// or [`Self::flush`] to flush all pending writes. #[async_trait::async_trait] pub trait DbWriteOps { /// The transaction type returned by [`Self::begin`]. Stub @@ -220,6 +225,9 @@ pub trait DbWriteOps { /// Write a value into the database with default `PutOptions` and /// `WriteOptions`. /// + /// This method does not wait for durability. See [`DbWriteOps`] for + /// details. + /// /// ## Arguments /// - `key`: the key to write /// - `value`: the value to write @@ -238,6 +246,9 @@ pub trait DbWriteOps { /// Write a value into the database with custom `PutOptions` and /// `WriteOptions`. /// + /// This method does not wait for durability. See [`DbWriteOps`] for + /// details. + /// /// ## Arguments /// - `key`: the key to write /// - `value`: the value to write @@ -259,6 +270,9 @@ pub trait DbWriteOps { /// Delete a key from the database with default `WriteOptions`. /// + /// This method does not wait for durability. See [`DbWriteOps`] for + /// details. + /// /// ## Arguments /// - `key`: the key to delete /// @@ -271,6 +285,9 @@ pub trait DbWriteOps { /// Delete a key from the database with custom `WriteOptions`. /// + /// This method does not wait for durability. See [`DbWriteOps`] for + /// details. + /// /// ## Arguments /// - `key`: the key to delete /// - `options`: the write options to use @@ -286,6 +303,9 @@ pub trait DbWriteOps { /// Merge a value into the database with default `MergeOptions` and /// `WriteOptions`. /// + /// This method does not wait for durability. See [`DbWriteOps`] for + /// details. + /// /// Merge operations allow applications to bypass the traditional /// read/modify/write cycle by expressing partial updates using an /// associative operator. The merge operator must be configured when @@ -315,6 +335,9 @@ pub trait DbWriteOps { /// Merge a value into the database with custom `MergeOptions` and /// `WriteOptions`. /// + /// This method does not wait for durability. See [`DbWriteOps`] for + /// details. + /// /// ## Arguments /// - `key`: the key to merge into /// - `value`: the merge operand to apply @@ -337,6 +360,9 @@ pub trait DbWriteOps { /// Write a batch of put/delete operations atomically to the database. /// + /// This method does not wait for durability. See [`DbWriteOps`] for + /// details. + /// /// ## Arguments /// - `batch`: the batch of operations to write /// @@ -350,6 +376,9 @@ pub trait DbWriteOps { /// Write a batch of put/delete operations atomically to the database with /// custom `WriteOptions`. /// + /// This method does not wait for durability. See [`DbWriteOps`] for + /// details. + /// /// ## Arguments /// - `batch`: the batch of operations to write /// - `options`: the write options to use @@ -362,8 +391,8 @@ pub trait DbWriteOps { options: &WriteOptions, ) -> Result; - /// Flush in-memory writes to disk. This function blocks until the - /// in-memory data has been durably written to object storage. + /// Flush in-memory writes to object storage. This function blocks until + /// the in-memory data has been durably written. /// /// ## Errors /// - `Error`: if there was an error flushing the database. @@ -484,6 +513,10 @@ pub trait DbTransactionOps: DbReadOps { /// Commit the transaction with default `WriteOptions`. /// + /// A successful commit applies the write atomically but does not wait for + /// durability. Call [`WriteHandle::await_durable`] on the returned handle + /// when the result is `Some`. + /// /// ## Returns /// - `Ok(Some(WriteHandle))` if the commit is successful and there are /// writes in the batch. @@ -500,6 +533,11 @@ pub trait DbTransactionOps: DbReadOps { } /// Commit the transaction with custom `WriteOptions`. + /// + /// A successful commit applies the write atomically but does not wait for + /// durability. Call [`WriteHandle::await_durable`] on the returned handle + /// when the result is `Some`. + /// async fn commit_with_options( self, options: &WriteOptions, diff --git a/slatedb/src/test_utils.rs b/slatedb/src/test_utils.rs index df22e70099..6eacf16021 100644 --- a/slatedb/src/test_utils.rs +++ b/slatedb/src/test_utils.rs @@ -367,17 +367,23 @@ where pub(crate) async fn seed_database( db: &Db, table: &BTreeMap, - await_durable: bool, + wait_for_durability: bool, ) -> Result<(), crate::Error> { let put_options = PutOptions::default(); - let write_options = WriteOptions { - await_durable, - ..Default::default() - }; + let write_options = WriteOptions::default(); + let mut last_handle = None; for (key, value) in table.iter() { - db.put_with_options(key, value, &put_options, &write_options) - .await?; + last_handle = Some( + db.put_with_options(key, value, &put_options, &write_options) + .await?, + ); + } + + if wait_for_durability { + if let Some(handle) = last_handle { + handle.await_durable().await?; + } } Ok(()) diff --git a/slatedb/src/wal/mod.rs b/slatedb/src/wal/mod.rs index eff19febab..14b5d6eb7b 100644 --- a/slatedb/src/wal/mod.rs +++ b/slatedb/src/wal/mod.rs @@ -173,7 +173,7 @@ pub struct WalStatus { #[derive(Debug, Clone)] pub enum WalEvent { /// Emitted when a WAL file is durably flushed to storage. On receipt of this event, SlateDB - /// notifies write tasks blocked on [`crate::config::WriteOptions::await_durable`] + /// advances the durable sequence number and notifies durability waiters. WalFlushed(WalStatus), /// Emitted when the WAL has closed with the final wal status containing the closed reason WalClosed(WalStatus), diff --git a/slatedb/tests/db.rs b/slatedb/tests/db.rs index 2d82e717bd..86d3608da9 100644 --- a/slatedb/tests/db.rs +++ b/slatedb/tests/db.rs @@ -57,7 +57,6 @@ async fn test_replay_wal_then_write() { value.as_bytes(), &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -90,7 +89,6 @@ async fn test_replay_wal_then_write() { b"new_value", &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) @@ -244,7 +242,6 @@ async fn test_concurrent_writers_and_readers() { i.to_be_bytes().as_ref(), &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() }, ) diff --git a/slatedb/tests/prefix_filter.rs b/slatedb/tests/prefix_filter.rs index ed7ec109de..5f8eae0861 100644 --- a/slatedb/tests/prefix_filter.rs +++ b/slatedb/tests/prefix_filter.rs @@ -119,10 +119,7 @@ mod composite_filters { async fn write_sample_data(db: &Db) { let put = PutOptions::default(); - let write = WriteOptions { - await_durable: false, - seqnum: 0, - }; + let write = WriteOptions::default(); // Write each batch in its own SST so multiple SSTs participate in the // read path and the filter has something to actually skip. for batch in [SAMPLE_USERS, SAMPLE_NON_USERS] { @@ -299,10 +296,7 @@ mod subrange { .expect("failed to build db"); let put = PutOptions::default(); - let write = WriteOptions { - await_durable: false, - seqnum: 0, - }; + let write = WriteOptions::default(); let ssts: &[&[&[u8]]] = &[ &[b"aaa1", b"ccc1"], // sandwich &[b"bbb1", b"bbb2", b"bbb3", b"bbb4"], @@ -493,10 +487,7 @@ mod empty_prefix_filter { let db = open_db(store.clone(), recorder.clone()).await; let put = PutOptions::default(); - let write = WriteOptions { - await_durable: false, - seqnum: 0, - }; + let write = WriteOptions::default(); for key in [b"a".as_slice(), b"b".as_slice()] { db.put_with_options(key, b"v", &put, &write) .await @@ -583,10 +574,7 @@ mod prop_test { async fn write_keys(db: &Db, keys: &[Vec]) { let put_opts = PutOptions::default(); - let write_opts = WriteOptions { - await_durable: false, - seqnum: 0, - }; + let write_opts = WriteOptions::default(); for (i, key) in keys.iter().enumerate() { let value = format!("v{}", i).into_bytes(); db.put_with_options(key, &value, &put_opts, &write_opts) diff --git a/website/src/content/docs/docs/design/writes.mdx b/website/src/content/docs/docs/design/writes.mdx index af15a2318a..f99eb442e0 100644 --- a/website/src/content/docs/docs/design/writes.mdx +++ b/website/src/content/docs/docs/design/writes.mdx @@ -2,7 +2,7 @@ title: Writes --- -Writes are moved to a background task as quickly as possible to prevent blocking the client. By default, `put()`, `write()`, and `delete()` use `WriteOptions::default()`, so calls wait for the write to become durable before returning. If you want lower latency and can tolerate losing in-flight data, set `await_durable` to `false`. You can also call `flush()` explicitly. The synchronous flow is as follows: +Writes are moved to a background task as quickly as possible to prevent blocking the client. `put()`, `write()`, and `delete()` return a `WriteHandle` after the write reaches the in-memory WAL and MemTable. Call `handle.await_durable().await` to wait for that write to reach object storage, or call `flush()` explicitly. The synchronous flow is as follows: 1. A `put()`, `write()`, or `delete()` call is made on the client. 2. The key/value pair is written to the mutable, in-memory WAL table. @@ -10,8 +10,8 @@ Writes are moved to a background task as quickly as possible to prevent blocking The following asynchronous flows occur: -- The WAL flusher periodically checks if the WAL table is full. If it is, it freezes the mutable WAL table and triggers an asynchronous write to object storage. A notification is then sent to clients that wrote with `await_durable` set to `true`. -- The MemTable flusher periodically checks if the MemTable is full. If it is, it freezes the mutable MemTable and triggers an asynchronous write to object storage. A notification is then sent to clients that wrote with `await_durable` set to `true` and `wal_enabled` set to `false`. +- The WAL flusher periodically checks if the WAL table is full. If it is, it freezes the mutable WAL table and triggers an asynchronous write to object storage. Durability waiters are notified when the durable sequence number advances. +- The MemTable flusher periodically checks if the MemTable is full. If it is, it freezes the mutable MemTable and triggers an asynchronous write to object storage. With the WAL disabled, this advances the durable sequence number and notifies durability waiters. Below is a diagram illustrating the high-level flow of a write in SlateDB: @@ -72,4 +72,4 @@ To avoid this, funnel all writes that use `seqnum` through a single ordering age - **Non-contiguous seqnums.** Auto-assigned seqnums are not guaranteed to be strictly contiguous either (the memtable flusher tolerates gaps), so user-supplied jumps are consistent with existing behavior. Anything that relies on dense seqnums — don't. - **Recovery.** On restart, SlateDB recovers the max seqnum from the manifest and any unflushed WAL. If you continue assigning seqnums from your external log, make sure the next value is above whatever SlateDB recovered, or your first post-restart write will be rejected. - **Transactions.** Conflict detection still works on user-supplied seqnums — the read/write set is tracked against the snapshot's seqnum and the commit seqnum, regardless of who picked them. Just keep the monotonic-increase invariant. -- **`await_durable`.** `seqnum` is independent of the `await_durable` flag; both can be set on the same `WriteOptions`. The oracle is advanced as soon as the batch is processed by the write loop, before durability is confirmed. The seqno the oracle tracks is still bumped, so if the write fails the seqno is still consumed. +- **Durability.** The oracle is advanced as soon as the batch is processed by the write loop, before durability is confirmed. The seqno is still consumed if a later durability wait fails. Use the returned `WriteHandle::await_durable()` when confirmation is required. diff --git a/website/src/content/docs/docs/get-started/faq.mdx b/website/src/content/docs/docs/get-started/faq.mdx index a33aeff719..a2a9ff7cfe 100644 --- a/website/src/content/docs/docs/get-started/faq.mdx +++ b/website/src/content/docs/docs/get-started/faq.mdx @@ -59,7 +59,7 @@ SlateDB also offers some unique features like the ability to create snapshot clo Any in-flight data that hasn't yet been flushed to object storage will be lost. -To prevent data loss, SlateDB's `put()` API will block until the data has been flushed to object storage. Client processes can block until their data has been durably written. Blocking can be disabled with [`WriteOptions`](https://docs.rs/slatedb/latest/slatedb/config/struct.WriteOptions.html) for clients that don't need this durability guarantee. +SlateDB's write APIs return after updating the in-memory WAL and MemTable. Call `handle.await_durable().await` on the returned [`WriteHandle`](https://docs.rs/slatedb/latest/slatedb/struct.WriteHandle.html) to wait for one write to reach object storage, or call `db.flush().await` to flush all pending writes. ## Does SlateDB support column families? From 287a36ebc8c4383a639d6564df6e763ad4e7b46c Mon Sep 17 00:00:00 2001 From: bowenli86 Date: Thu, 30 Jul 2026 11:11:40 -0700 Subject: [PATCH 07/65] test(dst): vary scan options (#1952) --- slatedb-dst/src/actors/workload.rs | 28 ++++++++++++++++++++++------ slatedb-dst/src/utils.rs | 24 +++++++++++++++++++++--- 2 files changed, 43 insertions(+), 9 deletions(-) diff --git a/slatedb-dst/src/actors/workload.rs b/slatedb-dst/src/actors/workload.rs index 6634717210..52e1615aae 100644 --- a/slatedb-dst/src/actors/workload.rs +++ b/slatedb-dst/src/actors/workload.rs @@ -5,13 +5,11 @@ use async_trait::async_trait; use bytes::Bytes; use log::info; use rand::RngCore; -use slatedb::config::{ - DurabilityLevel, MergeOptions, PutOptions, ReadOptions, ScanOptions, WriteOptions, -}; -use slatedb::{Error, MergeOperator, MergeOperatorError, WriteBatch}; +use slatedb::config::{DurabilityLevel, MergeOptions, PutOptions, ReadOptions, WriteOptions}; +use slatedb::{Error, IterationOrder, MergeOperator, MergeOperatorError, WriteBatch}; use tracing::instrument; -use crate::{Actor, ActorCtx}; +use crate::{utils::build_scan_options, Actor, ActorCtx}; use super::PROGRESS_LOG_INTERVAL; @@ -350,12 +348,13 @@ async fn verify_scan( read_durability: DurabilityLevel, observed: &mut BTreeMap, ) -> Result<(), Error> { - let scan_options = ScanOptions::new().with_durability_filter(read_durability); + let scan_options = build_scan_options(ctx.rand(), read_durability); let mut iter = ctx .db() .scan_prefix_with_options(key_prefix.as_bytes(), .., &scan_options) .await?; let mut seen = BTreeSet::new(); + let mut previous_key: Option = None; while let Some(key_value) = iter.next().await? { assert!( @@ -374,6 +373,23 @@ async fn verify_scan( key_prefix, String::from_utf8_lossy(key_value.key.as_ref()).into_owned(), ); + if let Some(previous_key) = previous_key.as_ref() { + let is_ordered = match scan_options.order { + IterationOrder::Ascending => previous_key < &key_value.key, + IterationOrder::Descending => previous_key > &key_value.key, + }; + assert!( + is_ordered, + "workload scan returned keys out of order [name={}, step={}, key_prefix={}, order={:?}, previous_key={}, key={}]", + ctx.name(), + step, + key_prefix, + scan_options.order, + String::from_utf8_lossy(previous_key.as_ref()), + String::from_utf8_lossy(key_value.key.as_ref()), + ); + } + previous_key = Some(key_value.key.clone()); observe_present(ctx, step, &key_value.key, &key_value.value, observed); } diff --git a/slatedb-dst/src/utils.rs b/slatedb-dst/src/utils.rs index 69c36de833..3bb05f3f41 100644 --- a/slatedb-dst/src/utils.rs +++ b/slatedb-dst/src/utils.rs @@ -4,11 +4,11 @@ use std::time::Duration; use rand::Rng; use slatedb::config::{ - CompactionWorkerOptions, CompactorOptions, CompressionCodec, DbReaderOptions, + CompactionWorkerOptions, CompactorOptions, CompressionCodec, DbReaderOptions, DurabilityLevel, GarbageCollectorDirectoryOptions, GarbageCollectorOptions, GarbageCollectorScheduleOptions, - SizeTieredCompactionSchedulerOptions, + ScanOptions, SizeTieredCompactionSchedulerOptions, }; -use slatedb::{DbRand, Settings}; +use slatedb::{DbRand, IterationOrder, Settings}; use tracing_subscriber::fmt::format::FmtSpan; use tracing_subscriber::EnvFilter; @@ -89,6 +89,24 @@ pub fn build_reader_options(rand: &DbRand) -> DbReaderOptions { } } +/// Builds randomized deterministic scan options for DST scenarios. +pub fn build_scan_options(rand: &DbRand, read_durability: DurabilityLevel) -> ScanOptions { + let mut rng = rand.rng(); + let read_ahead_options = [1, 4 * 1024, 64 * 1024, MIB_1]; + let order = if rng.random_bool(0.5) { + IterationOrder::Ascending + } else { + IterationOrder::Descending + }; + + ScanOptions::new() + .with_durability_filter(read_durability) + .with_read_ahead_bytes(read_ahead_options[rng.random_range(0..read_ahead_options.len())]) + .with_cache_blocks(rng.random_bool(0.5)) + .with_max_fetch_tasks(rng.random_range(1..=4)) + .with_order(order) +} + /// Builds randomized deterministic compactor options for DST scenarios. pub fn build_settings_compactor(rng: &mut impl Rng) -> CompactorOptions { let min_compaction_sources = rng.random_range(2..=4); From 2caad8723a1405a93366dcf595484b58df45a2ba Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?S=C3=B8ren=20Bramer=20Schmidt?= Date: Fri, 31 Jul 2026 05:33:37 +0700 Subject: [PATCH 08/65] Add cooperative yield points to SST build and compaction merge loops (#1964) --- slatedb/src/sst_builder.rs | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/slatedb/src/sst_builder.rs b/slatedb/src/sst_builder.rs index edc5853e8b..b26bfbca43 100644 --- a/slatedb/src/sst_builder.rs +++ b/slatedb/src/sst_builder.rs @@ -322,6 +322,10 @@ impl EncodedSsTableBuilder { self.blocks.push_back(block); self.first_key = None; + // Block encoding (compression/block transformer) is CPU-heavy. + // Give runtime a chance to run other tasks after each block. + tokio::task::yield_now().await; + Ok(Some(block_size)) } From 1b527bd26847fee1dc9a9a52e2e6c800f68a9fb1 Mon Sep 17 00:00:00 2001 From: Chris Date: Thu, 30 Jul 2026 15:52:03 -0700 Subject: [PATCH 09/65] Increase default GC interval to 10 minutes (#1993) --- slatedb/src/config.rs | 2 +- slatedb/src/garbage_collector.rs | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/slatedb/src/config.rs b/slatedb/src/config.rs index 83bd501651..b3bfd74948 100644 --- a/slatedb/src/config.rs +++ b/slatedb/src/config.rs @@ -1551,7 +1551,7 @@ impl GarbageCollectorOptions { /// Default options for the garbage collector for a directory. /// -/// By default, the garbage collector will run every minute and deletes files +/// By default, the garbage collector will run every 10 minutes and deletes files /// that are at least 5 minutes old. impl Default for GarbageCollectorDirectoryOptions { fn default() -> Self { diff --git a/slatedb/src/garbage_collector.rs b/slatedb/src/garbage_collector.rs index 426e235dcd..5677823af9 100644 --- a/slatedb/src/garbage_collector.rs +++ b/slatedb/src/garbage_collector.rs @@ -53,7 +53,7 @@ mod wal_gc; pub use filter::GcFilter; pub(crate) const DEFAULT_MIN_AGE: Duration = Duration::from_secs(300); -pub(crate) const DEFAULT_INTERVAL: Duration = Duration::from_secs(60); +pub(crate) const DEFAULT_INTERVAL: Duration = Duration::from_secs(600); pub(crate) const GC_TASK_NAME: &str = "garbage_collector"; /// Maximum number of concurrent object-store deletes issued by a GC task's /// deletion pass. Deletes are independent single-object operations, so a small From c7e9cafdae1a80c602c04f8a29b23fdf9f6f5ab2 Mon Sep 17 00:00:00 2001 From: nomiero Date: Thu, 30 Jul 2026 16:00:09 -0700 Subject: [PATCH 10/65] Remove unnecessary SR clones in scan APIs (#1994) --- bindings/uniffi/src/types.rs | 2 +- slatedb/Cargo.toml | 4 + .../scan_prefix_large_sorted_run_bench.rs | 242 ++++++++++++++++++ slatedb/src/compaction_worker.rs | 13 +- slatedb/src/compactor.rs | 212 +++++++-------- slatedb/src/compactor_executor.rs | 49 ++-- slatedb/src/compactor_state.rs | 100 ++------ slatedb/src/db.rs | 6 +- slatedb/src/db_reader.rs | 5 +- slatedb/src/db_state.rs | 59 ++++- slatedb/src/flatbuffer_types.rs | 100 ++++---- slatedb/src/garbage_collector.rs | 51 ++-- slatedb/src/garbage_collector/compacted_gc.rs | 7 +- slatedb/src/manifest/mod.rs | 137 +++++----- .../src/memtable_flusher/manifest_writer.rs | 2 +- slatedb/src/reader.rs | 16 +- slatedb/src/size_tiered_compaction.rs | 5 +- slatedb/src/sorted_run_iterator.rs | 47 ++-- slatedb/src/subcompaction.rs | 2 +- slatedb/src/test_utils.rs | 5 +- slatedb/src/utils.rs | 20 +- 21 files changed, 618 insertions(+), 466 deletions(-) create mode 100644 slatedb/benches/scan_prefix_large_sorted_run_bench.rs diff --git a/bindings/uniffi/src/types.rs b/bindings/uniffi/src/types.rs index 640f59e2b6..b426454915 100644 --- a/bindings/uniffi/src/types.rs +++ b/bindings/uniffi/src/types.rs @@ -814,7 +814,7 @@ impl From<&CoreSortedRun> for SortedRun { fn from(value: &CoreSortedRun) -> Self { Self { id: value.id, - sst_views: value.sst_views.iter().map(SsTableView::from).collect(), + sst_views: value.sst_views().iter().map(SsTableView::from).collect(), estimated_size_bytes: value.estimate_size(), } } diff --git a/slatedb/Cargo.toml b/slatedb/Cargo.toml index a8ddd80d6e..6fa88ac174 100644 --- a/slatedb/Cargo.toml +++ b/slatedb/Cargo.toml @@ -145,6 +145,10 @@ harness = false name = "scan_prefix_bench" harness = false +[[bench]] +name = "scan_prefix_large_sorted_run_bench" +harness = false + [[bench]] name = "block_iterator_v2" harness = false diff --git a/slatedb/benches/scan_prefix_large_sorted_run_bench.rs b/slatedb/benches/scan_prefix_large_sorted_run_bench.rs new file mode 100644 index 0000000000..f330ef9745 --- /dev/null +++ b/slatedb/benches/scan_prefix_large_sorted_run_bench.rs @@ -0,0 +1,242 @@ +// our microbenchmarks use pprof, but it doesn't work on windows +#![cfg(not(windows))] + +//! Measures how much of a prefix scan's setup cost comes from handing a +//! `SortedRun` to a scan iterator when the run holds far more SSTs than the +//! query range covers. +//! +//! Fixture: one sorted run of 1000 SSTs, one key per SST, disjoint and ordered +//! key ranges, with no memtable or L0 data left behind. Keys are `sr/000` +//! through `sr/999`, so each shorter prefix selects ten times as many SSTs. + +use std::collections::HashMap; +use std::sync::Arc; +use std::time::Duration; + +use bytes::Bytes; +use chrono::TimeDelta; +use criterion::{criterion_group, criterion_main, BenchmarkId, Criterion}; +use object_store::memory::InMemory; +use pprof::criterion::{Output, PProfProfiler}; +use slatedb::config::{ + CompactorOptions, DurabilityLevel, FlushOptions, FlushType, ScanOptions, Settings, +}; +use slatedb::Db; +use slatedb_common::clock::{DefaultSystemClock, SystemClock}; +use tokio::runtime::Runtime; + +/// Keys are `sr/000` through `sr/999`, one per SST. +const NUM_SSTS: usize = 1000; + +struct Case { + prefix: &'static str, + /// Keys under the prefix, which is also the number of rows a full prefix + /// scan returns. + keys: usize, + /// SST views `tables_covering_range` returns for the prefix range. + covering: usize, +} + +/// Each prefix drops one digit, so it selects ten times as many keys as the one +/// below it. +const CASES: [Case; 3] = [ + Case { + prefix: "sr/222", + keys: 1, + covering: 1, // the prefix is a single key, so it covers one SST view. + }, + Case { + prefix: "sr/22", + keys: 10, + covering: 11, // the prefix covers 11 SST views. + }, + Case { + prefix: "sr/2", + keys: 100, + covering: 101, // the prefix covers 101 SST views. + }, +]; + +fn make_key(idx: usize) -> Bytes { + Bytes::from(format!("sr/{idx:03}")) +} + +fn prefix_start(prefix: &'static str) -> Bytes { + Bytes::from_static(prefix.as_bytes()) +} + +/// Exclusive upper bound of the prefix range. Every prefix here ends in `2`, so +/// incrementing the last byte is enough. +fn prefix_end(prefix: &str) -> Bytes { + let mut end = prefix.as_bytes().to_vec(); + *end.last_mut().expect("prefix is non-empty") += 1; + Bytes::from(end) +} + +fn settings() -> Settings { + let mut scheduler_options = HashMap::new(); + // Fire exactly one compaction, and only once every L0 SST exists. + scheduler_options.insert("min_compaction_sources".to_string(), NUM_SSTS.to_string()); + scheduler_options.insert("max_compaction_sources".to_string(), NUM_SSTS.to_string()); + + Settings { + // Writers stall at these limits and the fixture parks NUM_SSTS in L0. + l0_max_ssts: NUM_SSTS + 16, + l0_max_ssts_per_key: NUM_SSTS + 16, + // One key per SST stays under the default min_filter_keys, so no SST + // carries a filter and every case is decided by key range overlap. + compactor_options: Some(CompactorOptions { + poll_interval: Duration::from_millis(50), + max_concurrent_compactions: 1, + // A trivial move preserves the one flush per SST topology. A + // rewriting compaction would choose its own output boundaries. + enable_trivial_move: true, + // No worker, so the coordinator either completes the compaction as + // a trivial move or the fixture assertion fails. + worker: None, + scheduler_options, + ..CompactorOptions::default() + }), + ..Settings::default() + } +} + +fn scan_options() -> ScanOptions { + ScanOptions { + cache_blocks: true, + durability_filter: DurabilityLevel::Remote, + ..ScanOptions::default() + } +} + +async fn populate(db: &Db) { + for idx in 0..NUM_SSTS { + let key = make_key(idx); + db.put(&key, &key).await.expect("put failed"); + db.flush_with_options(FlushOptions { + flush_type: FlushType::MemTable, + }) + .await + .expect("flush failed"); + } +} + +async fn await_single_sorted_run(db: &Db) { + let clock = DefaultSystemClock::new(); + let deadline = clock.now() + TimeDelta::seconds(300); + loop { + db.refresh_manifest() + .await + .expect("refresh_manifest failed"); + let manifest = db.manifest(); + let l0 = manifest.l0().len(); + let runs = manifest.compacted(); + let ssts = runs + .first() + .map_or(0, |run| run.tables_covering_range(..).len()); + if l0 == 0 && runs.len() == 1 && ssts == NUM_SSTS { + return; + } + assert!( + clock.now() < deadline, + "compaction never produced one sorted run of {NUM_SSTS} ssts (l0={l0}, runs={}, ssts={ssts})", + runs.len() + ); + clock.sleep(Duration::from_millis(50)).await; + } +} + +async fn count_prefix_rows(db: &Db, prefix: &'static str) -> usize { + let mut iter = db + .scan_prefix_with_options(prefix_start(prefix), .., &scan_options()) + .await + .expect("scan_prefix failed"); + let mut count = 0usize; + while iter.next().await.expect("iterator next failed").is_some() { + count += 1; + } + count +} + +async fn validate_fixture(db: &Db) { + let manifest = db.manifest(); + let run = manifest + .compacted() + .first() + .expect("fixture has one sorted run"); + for case in &CASES { + let covering = run + .tables_covering_range(prefix_start(case.prefix)..prefix_end(case.prefix)) + .len(); + assert_eq!( + covering, case.covering, + "prefix {} covers {covering} sst views", + case.prefix + ); + let rows = count_prefix_rows(db, case.prefix).await; + assert_eq!( + rows, case.keys, + "prefix {} scanned {rows} rows", + case.prefix + ); + } +} + +async fn build_fixture() -> Db { + let store = Arc::new(InMemory::new()); + let db = Db::builder("/bench/sorted_run", store) + .with_settings(settings()) + .build() + .await + .expect("failed to build db"); + populate(&db).await; + await_single_sorted_run(&db).await; + validate_fixture(&db).await; + db +} + +fn bench_scan_prefix_large_sorted_run(c: &mut Criterion) { + let runtime = Runtime::new().expect("failed to create runtime"); + let db = runtime.block_on(build_fixture()); + + let scan_opts = scan_options(); + let mut group = c.benchmark_group("scan_prefix_large_sorted_run"); + + for case in &CASES { + let label = format!("K={}", case.keys); + + group.bench_function(BenchmarkId::new("first_entry", &label), |b| { + b.to_async(&runtime).iter(|| async { + let mut iter = db + .scan_prefix_with_options(prefix_start(case.prefix), .., &scan_opts) + .await + .expect("scan_prefix failed"); + let entry = iter.next().await.expect("iterator next failed"); + assert!(entry.is_some()); + }); + }); + + group.bench_function(BenchmarkId::new("first_entry_by_recency", &label), |b| { + b.to_async(&runtime).iter(|| async { + let mut iter = db + .scan_prefix_by_recency_with_options(prefix_start(case.prefix), &scan_opts) + .await + .expect("scan_prefix_by_recency failed"); + let entry = iter.next_entry().await.expect("iterator next_entry failed"); + assert!(entry.is_some()); + }); + }); + } + + group.finish(); + runtime.block_on(async { db.close().await.expect("close failed") }); +} + +criterion_group! { + name = benches; + config = Criterion::default() + .with_profiler(PProfProfiler::new(100, Output::Protobuf)); + targets = bench_scan_prefix_large_sorted_run +} + +criterion_main!(benches); diff --git a/slatedb/src/compaction_worker.rs b/slatedb/src/compaction_worker.rs index 3b58f3790d..a330d31b39 100644 --- a/slatedb/src/compaction_worker.rs +++ b/slatedb/src/compaction_worker.rs @@ -616,7 +616,13 @@ impl CompactionWorkerHandler { let heartbeat_ms = self.clock.now().timestamp_millis() as u64; let updated = existing .with_status(CompactionStatus::Compacted) - .with_output_ssts(sorted_run.sst_views.iter().map(|v| v.sst.clone()).collect()) + .with_output_ssts( + sorted_run + .sst_views() + .iter() + .map(|v| v.sst.clone()) + .collect(), + ) .with_worker(Some(WorkerSpec::new(self.worker_id.clone(), heartbeat_ms))) .with_ctx(None); dirty.value.insert(updated); @@ -1114,10 +1120,7 @@ mod tests { ..SsTableInfo::default() }, ); - let sorted_run = SortedRun { - id: 0, - sst_views: vec![SsTableView::identity(output_handle.clone())], - }; + let sorted_run = SortedRun::new(0, [SsTableView::identity(output_handle.clone())]); fx.handler .handle_finished(id, Ok(sorted_run)) diff --git a/slatedb/src/compactor.rs b/slatedb/src/compactor.rs index 2312ce90ac..e26f29c42f 100644 --- a/slatedb/src/compactor.rs +++ b/slatedb/src/compactor.rs @@ -882,14 +882,13 @@ impl CompactorEventHandler { .spec() .destination() .expect("Compacted tiered compaction must have a destination SR id"); - let output_sr = SortedRun { - id: destination, - sst_views: compaction + let output_sr = SortedRun::new( + destination, + compaction .output_ssts() .iter() - .map(|sst| SsTableView::identity(sst.clone())) - .collect(), - }; + .map(|sst| SsTableView::identity(sst.clone())), + ); self.state_mut().finish_compaction(id, output_sr); manifest_changed = true; self.stats @@ -1705,7 +1704,7 @@ mod tests { // then: let db_state = db_state.expect("db was not compacted"); for run in db_state.tree.compacted.iter() { - for sst in run.sst_views.iter() { + for sst in run.sst_views() { let mut iter = SstIterator::new_borrowed_initialized( .., sst, @@ -1828,7 +1827,7 @@ mod tests { .tree .compacted .iter() - .flat_map(|sr| &sr.sst_views) + .flat_map(|sr| sr.sst_views().iter()) .collect(); assert_eq!(output_ssts.len(), 1); let view = output_ssts[0]; @@ -2355,7 +2354,7 @@ mod tests { .unwrap(); assert_eq!(db_state.tree.compacted.len(), 1); - let compacted = &db_state.tree.compacted.first().unwrap().sst_views; + let compacted = db_state.tree.compacted.first().unwrap().sst_views(); assert_eq!(compacted.len(), 1); let handle = compacted.first().unwrap(); @@ -2454,7 +2453,7 @@ mod tests { .unwrap(); assert_eq!(db_state.tree.compacted.len(), 1); - let compacted = &db_state.tree.compacted.first().unwrap().sst_views; + let compacted = db_state.tree.compacted.first().unwrap().sst_views(); assert_eq!(compacted.len(), 1); let handle = compacted.first().unwrap(); @@ -2589,7 +2588,7 @@ mod tests { // then: let db_state = db_state.expect("db was not compacted"); assert_eq!(db_state.tree.compacted.len(), 1); - let compacted = &db_state.tree.compacted.first().unwrap().sst_views; + let compacted = db_state.tree.compacted.first().unwrap().sst_views(); assert_eq!(compacted.len(), 1); let handle = compacted.first().unwrap(); @@ -2801,7 +2800,7 @@ mod tests { // then: let db_state = db_state.expect("db was not compacted"); assert_eq!(db_state.tree.compacted.len(), 1); - let compacted = &db_state.tree.compacted.first().unwrap().sst_views; + let compacted = db_state.tree.compacted.first().unwrap().sst_views(); assert_eq!(compacted.len(), 1); let handle = compacted.first().unwrap(); @@ -2918,7 +2917,7 @@ mod tests { // then: let db_state = db_state.expect("db was not compacted"); assert_eq!(db_state.tree.compacted.len(), 1); - let compacted = &db_state.tree.compacted.first().unwrap().sst_views; + let compacted = db_state.tree.compacted.first().unwrap().sst_views(); assert_eq!(compacted.len(), 1); let handle = compacted.first().unwrap(); @@ -3047,7 +3046,7 @@ mod tests { // then: let db_state = db_state.expect("db was not compacted"); assert_eq!(db_state.tree.compacted.len(), 1); - let compacted = &db_state.tree.compacted.first().unwrap().sst_views; + let compacted = db_state.tree.compacted.first().unwrap().sst_views(); assert_eq!(compacted.len(), 1); let handle = compacted.first().unwrap(); @@ -3151,7 +3150,7 @@ mod tests { assert_eq!(db_state.last_l0_clock_tick, 20); // then: the compacted SST should only contain the non-expired merge - let compacted = &db_state.tree.compacted.first().unwrap().sst_views; + let compacted = db_state.tree.compacted.first().unwrap().sst_views(); assert_eq!(compacted.len(), 1); let handle = compacted.first().unwrap(); @@ -3368,7 +3367,7 @@ mod tests { ); // The compacted sorted run should contain both merge operations separately - let compacted = &db_state.tree.compacted.first().unwrap().sst_views; + let compacted = db_state.tree.compacted.first().unwrap().sst_views(); assert_eq!(compacted.len(), 1); let handle = compacted.first().unwrap(); @@ -3483,7 +3482,7 @@ mod tests { "compaction should have occurred" ); - let compacted = &db_state.tree.compacted.first().unwrap().sst_views; + let compacted = db_state.tree.compacted.first().unwrap().sst_views(); assert_eq!(compacted.len(), 1); let handle = compacted.first().unwrap(); @@ -3611,7 +3610,7 @@ mod tests { let db_state = db_state.expect("db was not compacted"); assert!(db_state.tree.last_compacted_l0_sst_view_id.is_some()); assert_eq!(db_state.tree.compacted.len(), 1); - let compacted = &db_state.tree.compacted.first().unwrap().sst_views; + let compacted = db_state.tree.compacted.first().unwrap().sst_views(); assert_eq!(compacted.len(), 1); let handle = compacted.first().unwrap(); let mut iter = SstIterator::new_borrowed_initialized( @@ -3737,7 +3736,7 @@ mod tests { assert!(db_state.tree.last_compacted_l0_sst_view_id.is_some()); assert_eq!(db_state.tree.compacted.len(), 1); assert_eq!(db_state.last_l0_clock_tick, 70); - let compacted = &db_state.tree.compacted.first().unwrap().sst_views; + let compacted = db_state.tree.compacted.first().unwrap().sst_views(); assert_eq!(compacted.len(), 1); let handle = compacted.first().unwrap(); let mut iter = SstIterator::new_borrowed_initialized( @@ -3845,10 +3844,7 @@ mod tests { ..SsTableInfo::default() }, )); - let segment_sr = SortedRun { - id: 7, - sst_views: vec![segment_sr_view.clone()], - }; + let segment_sr = SortedRun::new(7, [segment_sr_view.clone()]); let mut core = ManifestCore::new(); core.segments = vec![Segment { @@ -4129,22 +4125,22 @@ mod tests { Arc::make_mut(&mut dirty.value.core.tree).l0 = VecDeque::from(vec![l0_view_newest, l0_view_oldest]); Arc::make_mut(&mut dirty.value.core.tree).compacted = vec![ - SortedRun { - id: 2, - sst_views: vec![SsTableView::identity(SsTableHandle::new( + SortedRun::new( + 2, + [SsTableView::identity(SsTableHandle::new( SsTableId::Compacted(Ulid::new()), SST_FORMAT_VERSION_LATEST, sr_info.clone(), ))], - }, - SortedRun { - id: 1, - sst_views: vec![SsTableView::identity(SsTableHandle::new( + ), + SortedRun::new( + 1, + [SsTableView::identity(SsTableHandle::new( SsTableId::Compacted(Ulid::new()), SST_FORMAT_VERSION_LATEST, sr_info.clone(), ))], - }, + ), ]; stored_manifest.update(dirty).await.unwrap(); @@ -4232,22 +4228,22 @@ mod tests { )); Arc::make_mut(&mut core.tree).l0 = VecDeque::from(vec![l0_view_first, l0_view_second]); Arc::make_mut(&mut core.tree).compacted = vec![ - SortedRun { - id: 5, - sst_views: vec![SsTableView::identity(SsTableHandle::new( + SortedRun::new( + 5, + [SsTableView::identity(SsTableHandle::new( SsTableId::Compacted(Ulid::from_parts(10, 0)), SST_FORMAT_VERSION_LATEST, sr_info.clone(), ))], - }, - SortedRun { - id: 2, - sst_views: vec![SsTableView::identity(SsTableHandle::new( + ), + SortedRun::new( + 2, + [SsTableView::identity(SsTableHandle::new( SsTableId::Compacted(Ulid::from_parts(11, 0)), SST_FORMAT_VERSION_LATEST, sr_info, ))], - }, + ), ]; let state = CompactorStateView { compactions: None, @@ -4329,22 +4325,22 @@ mod tests { last_compacted_l0_sst_id: None, l0: VecDeque::new(), compacted: vec![ - SortedRun { - id: 7, - sst_views: vec![SsTableView::identity(SsTableHandle::new( + SortedRun::new( + 7, + [SsTableView::identity(SsTableHandle::new( SsTableId::Compacted(Ulid::from_parts(70, 0)), SST_FORMAT_VERSION_LATEST, sr_info.clone(), ))], - }, - SortedRun { - id: 3, - sst_views: vec![SsTableView::identity(SsTableHandle::new( + ), + SortedRun::new( + 3, + [SsTableView::identity(SsTableHandle::new( SsTableId::Compacted(Ulid::from_parts(30, 0)), SST_FORMAT_VERSION_LATEST, sr_info, ))], - }, + ), ], }), }]; @@ -4384,14 +4380,14 @@ mod tests { first_entry: Some(Bytes::from_static(b"r")), ..SsTableInfo::default() }; - Arc::make_mut(&mut core.tree).compacted = vec![SortedRun { - id: 9, - sst_views: vec![SsTableView::identity(SsTableHandle::new( + Arc::make_mut(&mut core.tree).compacted = vec![SortedRun::new( + 9, + [SsTableView::identity(SsTableHandle::new( SsTableId::Compacted(Ulid::from_parts(90, 0)), SST_FORMAT_VERSION_LATEST, sr_info, ))], - }]; + )]; let state = CompactorStateView { compactions: None, manifest: VersionedManifest::from_manifest(0, Manifest::initial(core)), @@ -4420,14 +4416,14 @@ mod tests { first_entry: Some(Bytes::from_static(b"a")), ..SsTableInfo::default() }; - Arc::make_mut(&mut core.tree).compacted = vec![SortedRun { - id: 4, - sst_views: vec![SsTableView::identity(SsTableHandle::new( + Arc::make_mut(&mut core.tree).compacted = vec![SortedRun::new( + 4, + [SsTableView::identity(SsTableHandle::new( SsTableId::Compacted(Ulid::from_parts(40, 0)), SST_FORMAT_VERSION_LATEST, sr_info, ))], - }]; + )]; let state = CompactorStateView { compactions: None, manifest: VersionedManifest::from_manifest(0, Manifest::initial(core)), @@ -4459,22 +4455,22 @@ mod tests { ..SsTableInfo::default() }; Arc::make_mut(&mut core.tree).compacted = vec![ - SortedRun { - id: 8, - sst_views: vec![SsTableView::identity(SsTableHandle::new( + SortedRun::new( + 8, + [SsTableView::identity(SsTableHandle::new( SsTableId::Compacted(Ulid::from_parts(80, 0)), SST_FORMAT_VERSION_LATEST, info.clone(), ))], - }, - SortedRun { - id: 4, - sst_views: vec![SsTableView::identity(SsTableHandle::new( + ), + SortedRun::new( + 4, + [SsTableView::identity(SsTableHandle::new( SsTableId::Compacted(Ulid::from_parts(40, 0)), SST_FORMAT_VERSION_LATEST, info.clone(), ))], - }, + ), ]; core.segment_extractor_name = Some("test".into()); core.segments = vec![ @@ -4484,14 +4480,14 @@ mod tests { last_compacted_l0_sst_view_id: None, last_compacted_l0_sst_id: None, l0: VecDeque::new(), - compacted: vec![SortedRun { - id: 3, - sst_views: vec![SsTableView::identity(SsTableHandle::new( + compacted: vec![SortedRun::new( + 3, + [SsTableView::identity(SsTableHandle::new( SsTableId::Compacted(Ulid::from_parts(30, 0)), SST_FORMAT_VERSION_LATEST, info.clone(), ))], - }], + )], }), }, Segment { @@ -4501,22 +4497,22 @@ mod tests { last_compacted_l0_sst_id: None, l0: VecDeque::new(), compacted: vec![ - SortedRun { - id: 9, - sst_views: vec![SsTableView::identity(SsTableHandle::new( + SortedRun::new( + 9, + [SsTableView::identity(SsTableHandle::new( SsTableId::Compacted(Ulid::from_parts(90, 0)), SST_FORMAT_VERSION_LATEST, info.clone(), ))], - }, - SortedRun { - id: 6, - sst_views: vec![SsTableView::identity(SsTableHandle::new( + ), + SortedRun::new( + 6, + [SsTableView::identity(SsTableHandle::new( SsTableId::Compacted(Ulid::from_parts(60, 0)), SST_FORMAT_VERSION_LATEST, info, ))], - }, + ), ], }), }, @@ -4560,14 +4556,14 @@ mod tests { first_entry: Some(Bytes::from_static(b"x")), ..SsTableInfo::default() }; - Arc::make_mut(&mut core.tree).compacted = vec![SortedRun { - id: 5, - sst_views: vec![SsTableView::identity(SsTableHandle::new( + Arc::make_mut(&mut core.tree).compacted = vec![SortedRun::new( + 5, + [SsTableView::identity(SsTableHandle::new( SsTableId::Compacted(Ulid::from_parts(50, 0)), SST_FORMAT_VERSION_LATEST, info.clone(), ))], - }]; + )]; core.segment_extractor_name = Some("test".into()); core.segments = vec![ // L0-only: still skipped because L0 SSTs are ineligible inputs. @@ -4883,7 +4879,9 @@ mod tests { let completed = compaction .clone() .with_status(CompactionStatus::Compacted) - .with_output_ssts(result.sst_views.iter().map(|v| v.sst.clone()).collect()) + .with_output_ssts( + result.sst_views().iter().map(|v| v.sst.clone()).collect(), + ) .with_ctx(None); dirty.value.insert(completed); match stored.update(dirty).await { @@ -4952,7 +4950,7 @@ mod tests { .compacted .first() .unwrap() - .sst_views + .sst_views() .iter() .map(|view| view.sst.id.unwrap_compacted_id()) .collect(); @@ -5127,10 +5125,8 @@ mod tests { .value .core; Arc::make_mut(&mut core.tree).l0 = VecDeque::from([l0.clone()]); - Arc::make_mut(&mut core.tree).compacted = vec![SortedRun { - id: 1, - sst_views: vec![sr_first.clone(), sr_last.clone()], - }]; + Arc::make_mut(&mut core.tree).compacted = + vec![SortedRun::new(1, [sr_first.clone(), sr_last.clone()])]; let compaction_id = Ulid::new(); fixture @@ -5164,7 +5160,7 @@ mod tests { assert_eq!(output.id, 2); assert_eq!( output - .sst_views + .sst_views() .iter() .map(|view| view.id) .collect::>(), @@ -5199,10 +5195,7 @@ mod tests { .value .core; Arc::make_mut(&mut core.tree).l0 = VecDeque::from([l0.clone()]); - Arc::make_mut(&mut core.tree).compacted = vec![SortedRun { - id: 1, - sst_views: vec![sr_first, sr_last], - }]; + Arc::make_mut(&mut core.tree).compacted = vec![SortedRun::new(1, [sr_first, sr_last])]; let compaction_id = Ulid::new(); fixture .handler @@ -5691,10 +5684,7 @@ mod tests { // Root tree holds SR(7) — the global max. The segment-targeted spec // below proposes dst=3, which is above the segment's local max (0) // but below the global max. - Arc::make_mut(&mut core.tree).compacted = vec![SortedRun { - id: 7, - sst_views: Vec::new(), - }]; + Arc::make_mut(&mut core.tree).compacted = vec![SortedRun::new(7, [])]; core.segments = vec![Segment { prefix: prefix.clone(), tree: Arc::new(LsmTreeState { @@ -5736,10 +5726,7 @@ mod tests { .manifest_mut_for_test() .value .core; - Arc::make_mut(&mut core.tree).compacted = vec![SortedRun { - id: 7, - sst_views: Vec::new(), - }]; + Arc::make_mut(&mut core.tree).compacted = vec![SortedRun::new(7, [])]; core.segments = vec![Segment { prefix: prefix.clone(), tree: Arc::new(LsmTreeState { @@ -5789,10 +5776,7 @@ mod tests { last_compacted_l0_sst_view_id: None, last_compacted_l0_sst_id: None, l0: VecDeque::from(vec![make_view(l0_view)]), - compacted: vec![SortedRun { - id: 7, - sst_views: Vec::new(), - }], + compacted: vec![SortedRun::new(7, [])], }), }]; @@ -5976,10 +5960,7 @@ mod tests { .manifest_mut_for_test() .value .core; - Arc::make_mut(&mut core.tree).compacted = vec![SortedRun { - id: 99, - sst_views: Vec::new(), - }]; + Arc::make_mut(&mut core.tree).compacted = vec![SortedRun::new(99, [])]; let prefix = Bytes::from_static(b"seg/"); core.segments = vec![Segment { prefix: prefix.clone(), @@ -6019,10 +6000,7 @@ mod tests { .core; // Place SR(7) in the root tree. The segment-targeted spec below uses 7 as // its destination but does not list it among its sources. - Arc::make_mut(&mut core.tree).compacted = vec![SortedRun { - id: 7, - sst_views: Vec::new(), - }]; + Arc::make_mut(&mut core.tree).compacted = vec![SortedRun::new(7, [])]; // Seed SR(99) into the segment so the source-existence check passes and // destination-overwrite is the rejection reason. let prefix = Bytes::from_static(b"seg/"); @@ -6032,10 +6010,7 @@ mod tests { last_compacted_l0_sst_view_id: None, last_compacted_l0_sst_id: None, l0: VecDeque::new(), - compacted: vec![SortedRun { - id: 99, - sst_views: Vec::new(), - }], + compacted: vec![SortedRun::new(99, [])], }), }]; @@ -6231,7 +6206,7 @@ mod tests { .tree .compacted .first() - .is_some_and(|sr| sr.sst_views.len() == 1) + .is_some_and(|sr| sr.sst_views().len() == 1) } /// If a clock is provided, it will be advanced the clock by 60 seconds on each iteration to @@ -6412,7 +6387,10 @@ mod tests { sr.is_some(), "output SR {destination} not found in manifest" ); - assert_eq!(sr.unwrap().sst_views.first().unwrap().sst.id, output_sst.id); + assert_eq!( + sr.unwrap().sst_views().first().unwrap().sst.id, + output_sst.id + ); // given: a Compacted SR0→SR1 compaction to validate the SR source path removed when not in L0 let sr1_output_sst = fake_output_sst(); @@ -6444,7 +6422,7 @@ mod tests { let sr1 = core2.tree.compacted.iter().find(|sr| sr.id == 1); assert!(sr1.is_some(), "SR 1 should exist"); assert_eq!( - sr1.unwrap().sst_views.first().unwrap().sst.id, + sr1.unwrap().sst_views().first().unwrap().sst.id, sr1_output_sst.id ); let stored2 = fixture diff --git a/slatedb/src/compactor_executor.rs b/slatedb/src/compactor_executor.rs index f52630569c..8292f6e8ec 100644 --- a/slatedb/src/compactor_executor.rs +++ b/slatedb/src/compactor_executor.rs @@ -765,16 +765,13 @@ impl TokioCompactionExecutorInner { ); return Err(SlateDBError::CompactorExecutorFailed); } - Ok(SortedRun { - id: destination, - sst_views: output_ssts - .into_iter() - .map(|sst| { - let id = self.rand.rng().gen_ulid(self.clock.as_ref()); - SsTableView::new(id, sst.clone()) - }) - .collect(), - }) + Ok(SortedRun::new( + destination, + output_ssts.into_iter().map(|sst| { + let id = self.rand.rng().gen_ulid(self.clock.as_ref()); + SsTableView::new(id, sst.clone()) + }), + )) } /// Runs the merge for one key range of a compaction job and returns the @@ -1522,10 +1519,10 @@ mod tests { sr_ssts.extend(ssts); all_entries.extend(entries.iter().cloned()); } - sorted_runs.push(SortedRun { - id: sr_id as u32, - sst_views: sr_ssts.into_iter().map(SsTableView::identity).collect(), - }); + sorted_runs.push(SortedRun::new( + sr_id as u32, + sr_ssts.into_iter().map(SsTableView::identity), + )); } } @@ -1749,7 +1746,7 @@ mod tests { .unwrap(); let mut expected_entries = Vec::new(); - for view in &full_run.sst_views { + for view in full_run.sst_views() { let mut iter = SstIterator::new( SstView::Owned( Box::new(SsTableView::identity(view.sst.clone())), @@ -1801,7 +1798,7 @@ mod tests { .unwrap(); let mut resumed_entries = Vec::new(); - for view in &resumed_run.sst_views { + for view in resumed_run.sst_views() { let mut iter = SstIterator::new( SstView::Owned( Box::new(SsTableView::identity(view.sst.clone())), @@ -1828,7 +1825,7 @@ mod tests { /// runs can be compared for byte-identical merged output. async fn read_run_entries(table_store: &Arc, run: &SortedRun) -> Vec { let mut entries = Vec::new(); - for sst in &run.sst_views { + for sst in run.sst_views() { let mut iter = SstIterator::new( SstView::Borrowed(sst, BytesRange::from(..)), table_store.clone(), @@ -2079,7 +2076,7 @@ mod tests { .sum(); assert_eq!( final_output, - split.sst_views.len(), + split.sst_views().len(), "final snapshot should capture every output SST" ); } @@ -2148,7 +2145,7 @@ mod tests { ); for sst in snapshot.iter().flat_map(|s| s.output_ssts()) { assert!( - resumed.sst_views.iter().any(|v| v.sst.id == sst.id), + resumed.sst_views().iter().any(|v| v.sst.id == sst.id), "previously recorded subcompaction output SST was not reused" ); } @@ -2159,7 +2156,7 @@ mod tests { if index == snapshots.len() - 1 { let recorded: usize = snapshot.iter().map(|s| s.output_ssts().len()).sum(); assert_eq!( - resumed.sst_views.len(), + resumed.sst_views().len(), recorded, "resuming a completed compaction must not produce new SSTs" ); @@ -2740,7 +2737,7 @@ mod tests { // then: multiple output SSTs were produced (proving real boundaries and // background closes ran) ... - let result_ssts = &result.sst_views; + let result_ssts = result.sst_views(); assert!( result_ssts.len() >= 2, "expected multiple output SSTs, got {}", @@ -2748,7 +2745,7 @@ mod tests { ); // ... and the merged output preserves every key in ascending order. let mut read_back = Vec::new(); - for view in result_ssts { + for view in result_ssts.iter() { let mut iter = SstIterator::new( SstView::Owned( Box::new(SsTableView::identity(view.sst.clone())), @@ -2922,8 +2919,8 @@ mod tests { .await .unwrap(); - assert_eq!(1, result.sst_views.len()); - let sst = result.sst_views[0].clone(); + assert_eq!(1, result.sst_views().len()); + let sst = result.sst_views()[0].clone(); let mut iter = SstIterator::new( SstView::Borrowed(&sst, BytesRange::from(..)), table_store.clone(), @@ -3061,8 +3058,8 @@ mod tests { let result = ctx.run_compaction(vec![l0], true, None).await.unwrap(); // Verify the output SST - assert_eq!(1, result.sst_views.len()); - let sst = result.sst_views[0].clone(); + assert_eq!(1, result.sst_views().len()); + let sst = result.sst_views()[0].clone(); let mut iter = SstIterator::new( SstView::Borrowed(&sst, BytesRange::from(..)), table_store.clone(), diff --git a/slatedb/src/compactor_state.rs b/slatedb/src/compactor_state.rs index ad51aa7b65..b9e42bb28d 100644 --- a/slatedb/src/compactor_state.rs +++ b/slatedb/src/compactor_state.rs @@ -526,8 +526,8 @@ impl Compaction { let mut sst_views = self.get_l0_sst_views(db_state); sst_views.extend( self.get_sorted_runs(db_state) - .into_iter() - .flat_map(|sr| sr.sst_views), + .iter() + .flat_map(|sr| sr.sst_views().iter().cloned()), ); sst_views.sort_by(|left, right| { left.compacted_effective_range() @@ -542,10 +542,7 @@ impl Compaction { .intersect(pair[1].compacted_effective_range()) .is_none() })) - .then_some(SortedRun { - id: destination, - sst_views, - }) + .then_some(SortedRun::new(destination, sst_views)) } /// The stable id (ULID) used to track this compaction across messages and attempts. @@ -1310,10 +1307,8 @@ mod tests { let sr_last = bounded_sst_view(3, b"z", b"z"); let mut db_state = ManifestCore::new(); Arc::make_mut(&mut db_state.tree).l0 = VecDeque::from([l0.clone()]); - Arc::make_mut(&mut db_state.tree).compacted = vec![SortedRun { - id: 1, - sst_views: vec![sr_first.clone(), sr_last.clone()], - }]; + Arc::make_mut(&mut db_state.tree).compacted = + vec![SortedRun::new(1, [sr_first.clone(), sr_last.clone()])]; let compaction = Compaction::new( Ulid::new(), CompactionSpec::new(vec![SstView(l0.id), SourceId::SortedRun(1)], 2), @@ -1326,7 +1321,7 @@ mod tests { assert_eq!(output.id, 2); assert_eq!( output - .sst_views + .sst_views() .iter() .map(|view| view.id) .collect::>(), @@ -1340,10 +1335,7 @@ mod tests { let sr_view = bounded_sst_view(2, b"m", b"z"); let mut db_state = ManifestCore::new(); Arc::make_mut(&mut db_state.tree).l0 = VecDeque::from([l0.clone()]); - Arc::make_mut(&mut db_state.tree).compacted = vec![SortedRun { - id: 1, - sst_views: vec![sr_view], - }]; + Arc::make_mut(&mut db_state.tree).compacted = vec![SortedRun::new(1, [sr_view])]; let compaction = Compaction::new( Ulid::new(), CompactionSpec::new(vec![SstView(l0.id), SourceId::SortedRun(1)], 2), @@ -1664,11 +1656,7 @@ mod tests { .expect("failed to add compaction"); // when: - let compacted_ssts = before_compaction.tree.l0.iter().cloned().collect(); - let sr = SortedRun { - id: 0, - sst_views: compacted_ssts, - }; + let sr = SortedRun::new(0, before_compaction.tree.l0.iter().cloned()); state.finish_compaction(compaction_id, sr.clone()); // then: @@ -1679,14 +1667,14 @@ mod tests { assert_eq!(state.db_state().tree.l0.len(), 0); assert_eq!(state.db_state().tree.compacted.len(), 1); assert_eq!(state.db_state().tree.compacted.first().unwrap().id, sr.id); - let expected_ids: Vec = sr.sst_views.iter().map(|h| h.sst.id).collect(); + let expected_ids: Vec = sr.sst_views().iter().map(|h| h.sst.id).collect(); let found_ids: Vec = state .db_state() .tree .compacted .first() .unwrap() - .sst_views + .sst_views() .iter() .map(|h| h.sst.id) .collect(); @@ -1728,10 +1716,7 @@ mod tests { .add_compaction(Compaction::new(compaction_id, spec)) .expect("failed to add compaction"); - let sr = SortedRun { - id: 0, - sst_views: before_compaction.tree.l0.iter().cloned().collect(), - }; + let sr = SortedRun::new(0, before_compaction.tree.l0.iter().cloned()); state.finish_compaction(compaction_id, sr); let external_dbs = &state.manifest().value.external_dbs; @@ -1762,11 +1747,7 @@ mod tests { .expect("failed to add compaction"); // when: - let compacted_ssts = before_compaction.tree.l0.iter().cloned().collect(); - let sr = SortedRun { - id: 0, - sst_views: compacted_ssts, - }; + let sr = SortedRun::new(0, before_compaction.tree.l0.iter().cloned()); state.finish_compaction(compaction_id, sr); // then: @@ -1851,10 +1832,7 @@ mod tests { .expect("failed to add compaction"); state.finish_compaction( compaction_id, - SortedRun { - id: 0, - sst_views: vec![original_l0s.back().unwrap().clone()], - }, + SortedRun::new(0, [original_l0s.back().unwrap().clone()]), ); // open a new db and write another l0 let db = build_db(os.clone(), rt.handle()); @@ -1918,10 +1896,7 @@ mod tests { .expect("failed to add compaction"); state.finish_compaction( compaction_id, - SortedRun { - id: 0, - sst_views: original_l0s.clone().into(), - }, + SortedRun::new(0, original_l0s.iter().cloned()), ); assert_eq!(state.db_state().tree.l0.len(), 0); // open a new db and write another l0 @@ -2212,7 +2187,7 @@ mod tests { fn sorted_run_to_description(sr: &SortedRun) -> SortedRunDescription { SortedRunDescription { id: sr.id, - ssts: sr.sst_views.iter().map(|h| h.sst.id).collect(), + ssts: sr.sst_views().iter().map(|h| h.sst.id).collect(), } } @@ -2250,14 +2225,8 @@ mod tests { // Add a named segment with two compacted SRs (ids 5 and 3, list-position // ordered newest-first as per ManifestCore conventions). let prefix = Bytes::from_static(b"hour=12/"); - let sr5 = SortedRun { - id: 5, - sst_views: Vec::new(), - }; - let sr3 = SortedRun { - id: 3, - sst_views: Vec::new(), - }; + let sr5 = SortedRun::new(5, []); + let sr3 = SortedRun::new(3, []); let segment = Segment { prefix: prefix.clone(), tree: Arc::new(LsmTreeState { @@ -2283,10 +2252,7 @@ mod tests { .expect("failed to add compaction"); // Finish the compaction with a fresh output SR. - let output = SortedRun { - id: 7, - sst_views: Vec::new(), - }; + let output = SortedRun::new(7, []); state.finish_compaction(compaction_id, output); // The segment's compacted list now holds only the new SR(7). @@ -2317,10 +2283,7 @@ mod tests { last_compacted_l0_sst_view_id: None, last_compacted_l0_sst_id: None, l0: VecDeque::new(), - compacted: vec![SortedRun { - id: 7, - sst_views: Vec::new(), - }], + compacted: vec![SortedRun::new(7, [])], }), }]; @@ -2333,10 +2296,7 @@ mod tests { // Segment dropped after submission, before finish. state.manifest.value.core.segments = Vec::new(); - let output = SortedRun { - id: 7, - sst_views: Vec::new(), - }; + let output = SortedRun::new(7, []); state.finish_compaction(compaction_id, output); let compaction = state @@ -2358,10 +2318,7 @@ mod tests { // Seed real sources so the source-isolation check passes for both // submissions. Root tree gets SR(99); segment "seg/" gets SR(100). - Arc::make_mut(&mut state.manifest.value.core.tree).compacted = vec![SortedRun { - id: 99, - sst_views: Vec::new(), - }]; + Arc::make_mut(&mut state.manifest.value.core.tree).compacted = vec![SortedRun::new(99, [])]; let prefix = Bytes::from_static(b"seg/"); state.manifest.value.core.segments = vec![Segment { prefix: prefix.clone(), @@ -2369,10 +2326,7 @@ mod tests { last_compacted_l0_sst_view_id: None, last_compacted_l0_sst_id: None, l0: VecDeque::new(), - compacted: vec![SortedRun { - id: 100, - sst_views: Vec::new(), - }], + compacted: vec![SortedRun::new(100, [])], }), }]; @@ -2401,10 +2355,7 @@ mod tests { // Seed SR(7) in the root tree so the source-isolation check passes // for the first submission; the spec rewrites that SR (destination=7). - Arc::make_mut(&mut state.manifest.value.core.tree).compacted = vec![SortedRun { - id: 7, - sst_views: Vec::new(), - }]; + Arc::make_mut(&mut state.manifest.value.core.tree).compacted = vec![SortedRun::new(7, [])]; let first_id = rand.rng().gen_ulid(system_clock.as_ref()); let first = CompactionSpec::new(vec![SourceId::SortedRun(7)], 7); @@ -2559,10 +2510,7 @@ mod tests { // Two L0s (newest first) and one SR in the segment. let l0_newer = drain_test_view(2); let l0_older = drain_test_view(1); - let sr = SortedRun { - id: 5, - sst_views: Vec::new(), - }; + let sr = SortedRun::new(5, []); let prefix = Bytes::from_static(b"hour=10/"); state.manifest.value.core.segments = vec![Segment { prefix: prefix.clone(), diff --git a/slatedb/src/db.rs b/slatedb/src/db.rs index e638eb0916..7bb64eed8a 100644 --- a/slatedb/src/db.rs +++ b/slatedb/src/db.rs @@ -9310,7 +9310,7 @@ mod tests { .tree .compacted .first() - .is_some_and(|sr| sr.sst_views.len() > 1) + .is_some_and(|sr| sr.sst_views().len() > 1) { break; } @@ -9466,7 +9466,7 @@ mod tests { .tree .compacted .first() - .is_some_and(|sr| sr.sst_views.len() > 1) + .is_some_and(|sr| sr.sst_views().len() > 1) { break; } @@ -11663,7 +11663,7 @@ mod tests { .manifest() .compacted() .iter() - .flat_map(|sr| sr.sst_views.iter()) + .flat_map(|sr| sr.sst_views().iter()) .map(|v| v.sst.id) .filter(|id| !l0_ids.contains(id)) .collect() diff --git a/slatedb/src/db_reader.rs b/slatedb/src/db_reader.rs index f7119e1ce6..12e4c21bb1 100644 --- a/slatedb/src/db_reader.rs +++ b/slatedb/src/db_reader.rs @@ -3135,10 +3135,7 @@ mod tests { last_compacted_l0_sst_view_id: None, last_compacted_l0_sst_id: None, l0: VecDeque::from(vec![view(1)]), - compacted: vec![SortedRun { - id: 0, - sst_views: vec![view(4)], - }], + compacted: vec![SortedRun::new(0, [view(4)])], }, )]; assert!( diff --git a/slatedb/src/db_state.rs b/slatedb/src/db_state.rs index dffa54812d..7e0b6dc856 100644 --- a/slatedb/src/db_state.rs +++ b/slatedb/src/db_state.rs @@ -492,10 +492,27 @@ pub struct SortedRun { /// The unique identifier for this sorted run. pub id: u32, /// The list of SSTable views in this sorted run. - pub sst_views: Vec, + /// + /// Held behind an `Arc` so cloning a `SortedRun` (e.g. per read in the + /// scan path) is a single refcount bump rather than a deep clone of every + /// view's `Bytes` handles. + sst_views: Arc<[SsTableView]>, } impl SortedRun { + /// Create a sorted run from an ordered collection of SSTable views. + pub fn new(id: u32, sst_views: impl IntoIterator) -> Self { + Self { + id, + sst_views: sst_views.into_iter().collect(), + } + } + + /// Return the ordered SSTable views in this sorted run. + pub fn sst_views(&self) -> &[SsTableView] { + &self.sst_views + } + /// Estimate the total size of all SSTables in this sorted run. pub fn estimate_size(&self) -> u64 { self.sst_views.iter().map(|sst| sst.estimate_size()).sum() @@ -647,12 +664,12 @@ impl SortedRun { &self.sst_views[matching_range] } - pub(crate) fn into_tables_covering_range( - mut self, - range: &BytesRange, - ) -> VecDeque { + pub(crate) fn into_tables_covering_range(self, range: &BytesRange) -> VecDeque { let matching_range = self.table_idx_covering_range(range); - self.sst_views.drain(matching_range).collect() + // `sst_views` is shared behind an `Arc`, so we clone only the few + // covering views rather than draining the whole run. The full slice + // is released with a single refcount decrement when `self` drops. + self.sst_views[matching_range].iter().cloned().collect() } } @@ -1141,6 +1158,14 @@ mod tests { let sorted_first_keys: BTreeSet = table_first_keys.into_iter().collect(); let sorted_run = create_sorted_run(0, &sorted_first_keys); let covering_tables = sorted_run.tables_covering_range(range.clone()); + let borrowed_ids: Vec<_> = covering_tables.iter().map(|view| view.id).collect(); + let owned_ids: Vec<_> = sorted_run + .clone() + .into_tables_covering_range(&range) + .iter() + .map(|view| view.id) + .collect(); + assert_eq!(owned_ids, borrowed_ids); let first_key = sorted_first_keys.first().unwrap().clone(); let range_start_key = test_utils::bound_as_option(range.start_bound()) @@ -1175,15 +1200,15 @@ mod tests { #[test] fn test_sorted_run_collect_tables_for_point_key() { - let sorted_run = SortedRun { - id: 0, - sst_views: vec![ + let sorted_run = SortedRun::new( + 0, + [ create_compacted_sst_view_with_bounds(b"a", Some(b"k")), create_compacted_sst_view_with_bounds(b"k", Some(b"k")), create_compacted_sst_view_with_bounds(b"k", Some(b"m")), create_compacted_sst_view_with_bounds(b"z", Some(b"z")), ], - }; + ); let covering_tables = sorted_run.tables_covering_point_key(b"k"); assert_eq!(covering_tables.len(), 3); @@ -1203,15 +1228,21 @@ mod tests { assert!(sorted_run.tables_covering_point_key(b"0").is_empty()); } + #[test] + fn test_sorted_run_clone_shares_sst_views() { + let sorted_run = SortedRun::new(0, [create_compacted_sst_view(Some(Bytes::from("a")))]); + let cloned = sorted_run.clone(); + + assert!(Arc::ptr_eq(&sorted_run.sst_views, &cloned.sst_views)); + assert_eq!(sorted_run.sst_views(), cloned.sst_views()); + } + fn create_sorted_run(id: u32, first_keys: &BTreeSet) -> SortedRun { let mut ssts = Vec::new(); for first_key in first_keys { ssts.push(create_compacted_sst_view(Some(first_key.clone()))); } - SortedRun { - id, - sst_views: ssts, - } + SortedRun::new(id, ssts) } fn create_compacted_sst_view(first_entry: Option) -> SsTableView { diff --git a/slatedb/src/flatbuffer_types.rs b/slatedb/src/flatbuffer_types.rs index 7863e29702..e23347239b 100644 --- a/slatedb/src/flatbuffer_types.rs +++ b/slatedb/src/flatbuffer_types.rs @@ -277,10 +277,7 @@ impl FlatBufferManifestCodec { manifest_sst.visible_range().map(Self::decode_bytes_range), )); } - compacted.push(db_state::SortedRun { - id: manifest_sr.id(), - sst_views: ssts, - }) + compacted.push(db_state::SortedRun::new(manifest_sr.id(), ssts)) } let checkpoints: Vec = manifest .checkpoints() @@ -477,11 +474,8 @@ impl FlatBufferManifestCodec { .ssts() .iter() .map(|view| Self::decode_compacted_sst_view(&view, sst_lookup)) - .collect::>()?; - Ok(db_state::SortedRun { - id: sr.id(), - sst_views: ssts, - }) + .collect::, _>>()?; + Ok(db_state::SortedRun::new(sr.id(), ssts)) }, ) .collect() @@ -955,7 +949,7 @@ impl<'b> DbFlatBufferBuilder<'b> { &mut self, sorted_run: &db_state::SortedRun, ) -> WIPOffset> { - let ssts = self.add_compacted_sst_views(sorted_run.sst_views.iter()); + let ssts = self.add_compacted_sst_views(sorted_run.sst_views().iter()); SortedRunV2::create( &mut self.builder, &SortedRunV2Args { @@ -1022,7 +1016,7 @@ impl<'b> DbFlatBufferBuilder<'b> { sorted_run: &db_state::SortedRun, ) -> WIPOffset> { let ssts: Vec> = sorted_run - .sst_views + .sst_views() .iter() .map(|view| self.add_compacted_sst_from_view(view)) .collect(); @@ -1290,7 +1284,7 @@ impl<'b> DbFlatBufferBuilder<'b> { } } for sr in tree.compacted.iter() { - for view in sr.sst_views.iter() { + for view in sr.sst_views() { if let SsTableId::Compacted(ulid) = view.sst.id { unique_ssts.entry(ulid).or_insert(&view.sst); } @@ -1676,21 +1670,21 @@ mod tests { new_sst_handle(b"a", Some(BytesRange::from_ref("c"..="d"))), ]); Arc::make_mut(&mut manifest.core.tree).compacted = vec![ - SortedRun { - id: 0, - sst_views: vec![ + SortedRun::new( + 0, + [ new_sst_handle(b"a", None), new_sst_handle(b"d", Some(BytesRange::from_ref("e".."f"))), ], - }, - SortedRun { - id: 0, - sst_views: vec![ + ), + SortedRun::new( + 0, + [ new_sst_handle(b"a", None), new_sst_handle(b"c", Some(BytesRange::from_ref("c"..))), new_sst_handle(b"d", Some(BytesRange::from_ref("e".."f"))), ], - }, + ), ]; let codec = FlatBufferManifestCodec {}; @@ -1726,11 +1720,11 @@ mod tests { let root = Arc::make_mut(&mut manifest.core.tree); root.l0 = (0..16).map(new_sst_view).collect(); root.compacted = (0..4) - .map(|run| SortedRun { - id: run as u32, - sst_views: (0..8) - .map(|offset| new_sst_view(100 + run * 8 + offset)) - .collect(), + .map(|run| { + SortedRun::new( + run as u32, + (0..8).map(|offset| new_sst_view(100 + run * 8 + offset)), + ) }) .collect(); manifest.core.segment_extractor_name = Some("test".to_string()); @@ -1740,12 +1734,10 @@ mod tests { l0: (0..4) .map(|offset| new_sst_view(1_000 + segment * 16 + offset)) .collect(), - compacted: vec![SortedRun { - id: 100 + segment as u32, - sst_views: (0..4) - .map(|offset| new_sst_view(2_000 + segment * 16 + offset)) - .collect(), - }], + compacted: vec![SortedRun::new( + 100 + segment as u32, + (0..4).map(|offset| new_sst_view(2_000 + segment * 16 + offset)), + )], ..Default::default() }; Segment { @@ -1926,10 +1918,7 @@ mod tests { last_compacted_l0_sst_view_id: None, last_compacted_l0_sst_id: None, l0: VecDeque::new(), - compacted: vec![SortedRun { - id: 0, - sst_views: vec![new_sst_view(), new_sst_view()], - }], + compacted: vec![SortedRun::new(0, [new_sst_view(), new_sst_view()])], }), }, Segment { @@ -1938,10 +1927,7 @@ mod tests { last_compacted_l0_sst_view_id: None, last_compacted_l0_sst_id: None, l0: VecDeque::from(vec![new_sst_view(), new_sst_view()]), - compacted: vec![SortedRun { - id: 1, - sst_views: vec![new_sst_view()], - }], + compacted: vec![SortedRun::new(1, [new_sst_view()])], }), }, ]; @@ -2323,9 +2309,9 @@ mod tests { ..Default::default() }, ))]); - Arc::make_mut(&mut manifest.core.tree).compacted = vec![SortedRun { - id: 1, - sst_views: vec![SsTableView::identity(SsTableHandle::new( + Arc::make_mut(&mut manifest.core.tree).compacted = vec![SortedRun::new( + 1, + [SsTableView::identity(SsTableHandle::new( SsTableId::Compacted(ulid::Ulid::new()), SST_FORMAT_VERSION_LATEST, SsTableInfo { @@ -2333,7 +2319,7 @@ mod tests { ..Default::default() }, ))], - }]; + )]; let codec = FlatBufferManifestCodec {}; // when: @@ -2346,7 +2332,7 @@ mod tests { SST_FORMAT_VERSION_LATEST ); assert_eq!( - decoded.core.tree.compacted[0].sst_views[0] + decoded.core.tree.compacted[0].sst_views()[0] .sst .format_version, SST_FORMAT_VERSION_LATEST @@ -2452,7 +2438,7 @@ mod tests { super::ORIGINAL_SST_FORMAT_VERSION ); assert_eq!( - decoded.core.tree.compacted[0].sst_views[0] + decoded.core.tree.compacted[0].sst_views()[0] .sst .format_version, super::ORIGINAL_SST_FORMAT_VERSION @@ -2774,20 +2760,20 @@ mod tests { new_view(b"b", Some(BytesRange::from_ref("c"..="d"))), ]); Arc::make_mut(&mut manifest.core.tree).compacted = vec![ - SortedRun { - id: 1, - sst_views: vec![ + SortedRun::new( + 1, + [ new_view(b"e", None), new_view(b"f", Some(BytesRange::from_ref("g".."h"))), ], - }, - SortedRun { - id: 2, - sst_views: vec![ + ), + SortedRun::new( + 2, + [ new_view(b"i", None), new_view(b"j", Some(BytesRange::from_ref("k"..))), ], - }, + ), ]; Arc::make_mut(&mut manifest.core.tree).last_compacted_l0_sst_view_id = Some(manifest.core.tree.l0[0].id); @@ -2848,9 +2834,9 @@ mod tests { ..Default::default() }, ))]); - Arc::make_mut(&mut manifest.core.tree).compacted = vec![SortedRun { - id: 1, - sst_views: vec![SsTableView::identity(SsTableHandle::new( + Arc::make_mut(&mut manifest.core.tree).compacted = vec![SortedRun::new( + 1, + [SsTableView::identity(SsTableHandle::new( SsTableId::Compacted(ulid::Ulid::new()), SST_FORMAT_VERSION_LATEST, SsTableInfo { @@ -2858,7 +2844,7 @@ mod tests { ..Default::default() }, ))], - }]; + )]; manifest.writer_epoch = 5; manifest.compactor_epoch = 3; diff --git a/slatedb/src/garbage_collector.rs b/slatedb/src/garbage_collector.rs index 5677823af9..3098d9b0fa 100644 --- a/slatedb/src/garbage_collector.rs +++ b/slatedb/src/garbage_collector.rs @@ -1449,14 +1449,16 @@ mod tests { .l0 .push_back(SsTableView::identity(active_expired_l0_sst_handle.clone())); // Dont' push inactive_expired_l0_sst_handle - Arc::make_mut(&mut state.tree).compacted.push(SortedRun { - id: 1, - // Don't add inactive_expired_sst_handle - sst_views: vec![ - SsTableView::identity(active_sst_handle.clone()), - SsTableView::identity(active_expired_sst_handle.clone()), - ], - }); + Arc::make_mut(&mut state.tree) + .compacted + .push(SortedRun::new( + 1, + // Don't add inactive_expired_sst_handle + [ + SsTableView::identity(active_sst_handle.clone()), + SsTableView::identity(active_expired_sst_handle.clone()), + ], + )); StoredManifest::create_new_db( manifest_store.clone(), state.clone(), @@ -1488,7 +1490,7 @@ mod tests { assert_eq!(current_manifest.manifest.core.tree.compacted.len(), 1); assert_eq!( current_manifest.manifest.core.tree.compacted[0] - .sst_views + .sst_views() .len(), 2 ); @@ -1525,7 +1527,7 @@ mod tests { assert_eq!(current_manifest.manifest.core.tree.compacted.len(), 1); assert_eq!( current_manifest.manifest.core.tree.compacted[0] - .sst_views + .sst_views() .len(), 2 ); @@ -1569,14 +1571,18 @@ mod tests { .push_back(SsTableView::identity( active_checkpoint_l0_sst_handle.clone(), )); - Arc::make_mut(&mut state.tree).compacted.push(SortedRun { - id: 1, - sst_views: vec![SsTableView::identity(active_sst_handle.clone())], - }); - Arc::make_mut(&mut state.tree).compacted.push(SortedRun { - id: 2, - sst_views: vec![SsTableView::identity(active_checkpoint_sst_handle.clone())], - }); + Arc::make_mut(&mut state.tree) + .compacted + .push(SortedRun::new( + 1, + [SsTableView::identity(active_sst_handle.clone())], + )); + Arc::make_mut(&mut state.tree) + .compacted + .push(SortedRun::new( + 2, + [SsTableView::identity(active_checkpoint_sst_handle.clone())], + )); let mut stored_manifest = StoredManifest::create_new_db( manifest_store.clone(), state.clone(), @@ -1756,7 +1762,7 @@ mod tests { } for sr in &manifest.core.tree.compacted { - for view in &sr.sst_views { + for view in sr.sst_views() { assert!(compacted_ssts.contains(&view.sst.id)); } } @@ -2221,10 +2227,9 @@ mod tests { Arc::make_mut(&mut state.tree) .l0 .push_back(SsTableView::identity(active_l0_handle)); - Arc::make_mut(&mut state.tree).compacted.push(SortedRun { - id: 1, - sst_views: vec![SsTableView::identity(active_handle)], - }); + Arc::make_mut(&mut state.tree) + .compacted + .push(SortedRun::new(1, [SsTableView::identity(active_handle)])); // inactive_expired_handle is NOT in manifest -> eligible for GC StoredManifest::create_new_db( manifest_store.clone(), diff --git a/slatedb/src/garbage_collector/compacted_gc.rs b/slatedb/src/garbage_collector/compacted_gc.rs index de1570f663..9009e91e1a 100644 --- a/slatedb/src/garbage_collector/compacted_gc.rs +++ b/slatedb/src/garbage_collector/compacted_gc.rs @@ -149,7 +149,7 @@ fn collect_active_ssts<'a>(manifests: impl Iterator) -> Has active.insert(view.sst.id); } for sr in tree.compacted.iter() { - for view in sr.sst_views.iter() { + for view in sr.sst_views() { active.insert(view.sst.id); } } @@ -741,10 +741,7 @@ mod tests { last_compacted_l0_sst_view_id: None, last_compacted_l0_sst_id: None, l0: VecDeque::from(vec![segment_l0.clone()]), - compacted: vec![SortedRun { - id: 0, - sst_views: vec![segment_sr.clone()], - }], + compacted: vec![SortedRun::new(0, [segment_sr.clone()])], }), }], ); diff --git a/slatedb/src/manifest/mod.rs b/slatedb/src/manifest/mod.rs index 1995d9e334..5091e23b59 100644 --- a/slatedb/src/manifest/mod.rs +++ b/slatedb/src/manifest/mod.rs @@ -74,7 +74,7 @@ impl LsmTreeState { + self .compacted .iter() - .map(|sr| sr.sst_views.len()) + .map(|sr| sr.sst_views().len()) .sum::() } @@ -554,7 +554,7 @@ impl ManifestCore { self.trees().flat_map(|tree| { tree.l0 .iter() - .chain(tree.compacted.iter().flat_map(|sr| sr.sst_views.iter())) + .chain(tree.compacted.iter().flat_map(|sr| sr.sst_views().iter())) }) } @@ -1151,12 +1151,9 @@ impl Manifest { let l0: VecDeque = Self::filter_view_handles(&tree.l0, true, range).into(); let mut sorted_runs_filtered = vec![]; for sr in &tree.compacted { - let sst_views = Self::filter_view_handles(&sr.sst_views, false, range); + let sst_views = Self::filter_view_handles(sr.sst_views().iter(), false, range); if !sst_views.is_empty() { - sorted_runs_filtered.push(SortedRun { - id: sr.id, - sst_views, - }); + sorted_runs_filtered.push(SortedRun::new(sr.id, sst_views)); } } tree.l0 = l0; @@ -2086,7 +2083,7 @@ mod tests { l0: writer_l0.clone(), compacted: vec![], }; - let compactor_compacted = vec![SortedRun { id: 42, sst_views: vec![] }]; + let compactor_compacted = vec![SortedRun::new(42, [])]; let compactor = LsmTreeState { last_compacted_l0_sst_view_id: last_view, last_compacted_l0_sst_id: last_sst, @@ -2469,13 +2466,8 @@ mod tests { } tree.last_compacted_l0_sst_view_id = Some(newest); self.next_sr_id += 1; - tree.compacted.insert( - 0, - SortedRun { - id: self.next_sr_id, - sst_views: sr_views, - }, - ); + tree.compacted + .insert(0, SortedRun::new(self.next_sr_id, sr_views)); } } @@ -2574,7 +2566,7 @@ mod tests { } for seg in &self.store { for sr in &seg.tree.compacted { - for view in &sr.sst_views { + for view in sr.sst_views() { assert!( self.flushed_l0s.contains(&view.id), "SR {} references L0 {} that was never flushed", @@ -2661,7 +2653,7 @@ mod tests { for seg in &self.store { let l0_ids: BTreeSet = seg.tree.l0.iter().map(|v| v.id).collect(); for sr in &seg.tree.compacted { - for view in &sr.sst_views { + for view in sr.sst_views() { assert!( !l0_ids.contains(&view.id), "L0 {} appears in both l0 list and SR {} of segment {:?}", @@ -2978,28 +2970,25 @@ mod tests { )); } for (idx, sorted_run) in manifest.sorted_runs.iter().enumerate() { - Arc::make_mut(&mut core.tree).compacted.push(SortedRun { - id: idx as u32, - sst_views: sorted_run - .iter() - .map(|entry| { - let sst_id = sst_id_fn(entry.sst_alias); - let view_id = sst_id.unwrap_compacted_id(); - SsTableView::new_projected( - view_id, - SsTableHandle::new( - sst_id, - SST_FORMAT_VERSION_LATEST, - SsTableInfo { - first_entry: Some(entry.first_entry.clone()), - ..SsTableInfo::default() - }, - ), - entry.visible_range.clone(), - ) - }) - .collect(), - }); + Arc::make_mut(&mut core.tree).compacted.push(SortedRun::new( + idx as u32, + sorted_run.iter().map(|entry| { + let sst_id = sst_id_fn(entry.sst_alias); + let view_id = sst_id.unwrap_compacted_id(); + SsTableView::new_projected( + view_id, + SsTableHandle::new( + sst_id, + SST_FORMAT_VERSION_LATEST, + SsTableInfo { + first_entry: Some(entry.first_entry.clone()), + ..SsTableInfo::default() + }, + ), + entry.visible_range.clone(), + ) + }), + )); } Manifest::initial(core) } @@ -3172,10 +3161,9 @@ mod tests { Arc::make_mut(&mut core.tree) .l0 .push_back(create_sst_view(live_l0, b"a")); - Arc::make_mut(&mut core.tree).compacted.push(SortedRun { - id: 0, - sst_views: vec![create_sst_view(live_compacted, b"b")], - }); + Arc::make_mut(&mut core.tree) + .compacted + .push(SortedRun::new(0, [create_sst_view(live_compacted, b"b")])); let mut manifest = Manifest::initial(core); manifest.external_dbs = vec![ @@ -3229,10 +3217,10 @@ mod tests { last_compacted_l0_sst_view_id: None, last_compacted_l0_sst_id: None, l0: VecDeque::from(vec![create_sst_view(segment_l0, b"seg/a")]), - compacted: vec![SortedRun { - id: 0, - sst_views: vec![create_sst_view(segment_compacted, b"seg/b")], - }], + compacted: vec![SortedRun::new( + 0, + [create_sst_view(segment_compacted, b"seg/b")], + )], }), }]; @@ -3276,9 +3264,9 @@ mod tests { visible_range: BytesRange, ) -> Manifest { let mut core = ManifestCore::new(); - Arc::make_mut(&mut core.tree).compacted.push(SortedRun { - id: 0, - sst_views: vec![SsTableView::new_projected( + Arc::make_mut(&mut core.tree).compacted.push(SortedRun::new( + 0, + [SsTableView::new_projected( sst_id.unwrap_compacted_id(), SsTableHandle::new( sst_id, @@ -3290,7 +3278,7 @@ mod tests { ), Some(visible_range), )], - }); + )); Manifest::initial(core) } @@ -3975,9 +3963,9 @@ mod tests { ), Some(visible_range.clone()), )]), - compacted: vec![SortedRun { - id: 0, // gets renumbered globally by the union - sst_views: vec![SsTableView::new_projected( + compacted: vec![SortedRun::new( + 0, // gets renumbered globally by the union + [SsTableView::new_projected( sr_sst.unwrap_compacted_id(), SsTableHandle::new( sr_sst, @@ -3989,7 +3977,7 @@ mod tests { ), Some(visible_range), )], - }], + )], }), }]; (Manifest::initial(core), l0_sst, sr_sst) @@ -4047,10 +4035,7 @@ mod tests { // entry has the highest id (matching the descending-id-by-list- // position convention). fn make_sr(id: u32) -> SortedRun { - SortedRun { - id, // intentionally collides across trees pre-renumber - sst_views: vec![], - } + SortedRun::new(id, []) // intentionally collides across trees pre-renumber } let mut core = ManifestCore::new(); @@ -4252,9 +4237,9 @@ mod tests { ), Some(range.clone()), )]), - compacted: vec![SortedRun { - id: 0, - sst_views: vec![SsTableView::new_projected( + compacted: vec![SortedRun::new( + 0, + [SsTableView::new_projected( sr_id.unwrap_compacted_id(), SsTableHandle::new( sr_id, @@ -4266,7 +4251,7 @@ mod tests { ), Some(range), )], - }], + )], }), } } @@ -4693,9 +4678,9 @@ mod tests { ), Some(BytesRange::from_ref("a".."d")), )]), - compacted: vec![SortedRun { - id: 0, - sst_views: vec![SsTableView::new_projected( + compacted: vec![SortedRun::new( + 0, + [SsTableView::new_projected( sr_a.unwrap_compacted_id(), SsTableHandle::new( sr_a, @@ -4707,7 +4692,7 @@ mod tests { ), Some(BytesRange::from_ref("a".."m")), )], - }], + )], }), }, Segment { @@ -4716,9 +4701,9 @@ mod tests { last_compacted_l0_sst_view_id: None, last_compacted_l0_sst_id: None, l0: VecDeque::new(), - compacted: vec![SortedRun { - id: 1, - sst_views: vec![SsTableView::new_projected( + compacted: vec![SortedRun::new( + 1, + [SsTableView::new_projected( sr_b.unwrap_compacted_id(), SsTableHandle::new( sr_b, @@ -4730,7 +4715,7 @@ mod tests { ), Some(BytesRange::from_ref("n".."z")), )], - }], + )], }), }, ]; @@ -4774,13 +4759,13 @@ mod tests { ) }; let mut core = ManifestCore::new(); - Arc::make_mut(&mut core.tree).compacted.push(SortedRun { - id: 0, - sst_views: vec![ + Arc::make_mut(&mut core.tree).compacted.push(SortedRun::new( + 0, + [ make(sst1, b"a", b"c", BytesRange::from_ref("a".."d")), make(sst2, b"m", b"p", BytesRange::from_ref("m".."q")), ], - }); + )); Manifest::initial(core) } @@ -4814,7 +4799,7 @@ mod tests { ) .unwrap(); assert_eq!(projected.core.tree.compacted.len(), 1); - assert_eq!(projected.core.tree.compacted[0].sst_views.len(), 1); + assert_eq!(projected.core.tree.compacted[0].sst_views().len(), 1); } #[test] diff --git a/slatedb/src/memtable_flusher/manifest_writer.rs b/slatedb/src/memtable_flusher/manifest_writer.rs index a9e4055a98..7e1c94557a 100644 --- a/slatedb/src/memtable_flusher/manifest_writer.rs +++ b/slatedb/src/memtable_flusher/manifest_writer.rs @@ -686,7 +686,7 @@ impl ManifestWriterHandler { let all_views = tree .l0 .iter() - .chain(tree.compacted.iter().flat_map(|run| run.sst_views.iter())); + .chain(tree.compacted.iter().flat_map(|run| run.sst_views().iter())); for view in all_views { sst_views += 1; // Dedupe by physical SST id: a range clone/rescale can project one SST into diff --git a/slatedb/src/reader.rs b/slatedb/src/reader.rs index ad4b95ac03..1b90816fc5 100644 --- a/slatedb/src/reader.rs +++ b/slatedb/src/reader.rs @@ -574,17 +574,11 @@ mod tests { } let sst_handle = self.build_sst(entries).await?; - // Find or create the sorted run - let tree = Arc::make_mut(&mut self.core.tree); - if let Some(sr) = tree.compacted.iter_mut().find(|sr| sr.id == sr_id) { - sr.sst_views.push(SsTableView::identity(sst_handle)); - } else { - let new_sr = SortedRun { - id: sr_id, - sst_views: vec![SsTableView::identity(sst_handle)], - }; - tree.compacted.push(new_sr); - } + // The fixture groups all entries for a run into one SST before + // calling this helper, so each run is constructed exactly once. + Arc::make_mut(&mut self.core.tree) + .compacted + .push(SortedRun::new(sr_id, [SsTableView::identity(sst_handle)])); Ok(()) } diff --git a/slatedb/src/size_tiered_compaction.rs b/slatedb/src/size_tiered_compaction.rs index 616ad5a2b6..52fd63d3b2 100644 --- a/slatedb/src/size_tiered_compaction.rs +++ b/slatedb/src/size_tiered_compaction.rs @@ -1011,10 +1011,7 @@ mod tests { fn create_sr(id: u32, sst_size: u64, num_ssts: usize) -> SortedRun { let ssts: Vec = (0..num_ssts).map(|_| create_sst_view(sst_size)).collect(); - SortedRun { - id, - sst_views: ssts, - } + SortedRun::new(id, ssts) } fn create_db_state(l0: VecDeque, srs: Vec) -> ManifestCore { diff --git a/slatedb/src/sorted_run_iterator.rs b/slatedb/src/sorted_run_iterator.rs index 8e7646393c..8976e494f6 100644 --- a/slatedb/src/sorted_run_iterator.rs +++ b/slatedb/src/sorted_run_iterator.rs @@ -300,10 +300,7 @@ mod tests { let encoded = builder.build().await.unwrap(); let id = SsTableId::Compacted(ulid::Ulid::new()); let handle = table_store.write_sst(&id, &encoded).await.unwrap(); - let sr = SortedRun { - id: 0, - sst_views: vec![SsTableView::identity(handle)], - }; + let sr = SortedRun::new(0, [SsTableView::identity(handle)]); let mut iter = SortedRunIterator::new_owned_initialized( .., @@ -363,13 +360,13 @@ mod tests { let encoded = builder.build().await.unwrap(); let id2 = SsTableId::Compacted(ulid::Ulid::new()); let handle2 = table_store.write_sst(&id2, &encoded).await.unwrap(); - let sr = SortedRun { - id: 0, - sst_views: vec![ + let sr = SortedRun::new( + 0, + [ SsTableView::identity(handle1), SsTableView::identity(handle2), ], - }; + ); let mut iter = SortedRunIterator::new_owned_initialized( .., @@ -435,9 +432,9 @@ mod tests { let encoded = builder.build().await.unwrap(); let id2 = SsTableId::Compacted(ulid::Ulid::new()); let handle2 = table_store.write_sst(&id2, &encoded).await.unwrap(); - let sr = SortedRun { - id: 0, - sst_views: vec![ + let sr = SortedRun::new( + 0, + [ SsTableView::new_projected( ulid::Ulid::new(), handle1, @@ -449,7 +446,7 @@ mod tests { Some(BytesRange::from_ref("key5".."key7")), ), ], - }; + ); // when: iterating the full range, then: only visible keys appear let mut iter = SortedRunIterator::new_borrowed_initialized( @@ -679,10 +676,7 @@ mod tests { ssts.push(SsTableView::identity(handle)); } - SortedRun { - id: 0, - sst_views: ssts, - } + SortedRun::new(0, ssts) } async fn build_sr_with_ssts( @@ -703,10 +697,7 @@ mod tests { let sst = writer.close().await.unwrap(); ssts.push(SsTableView::identity(sst)); } - SortedRun { - id: 0, - sst_views: ssts, - } + SortedRun::new(0, ssts) } mod mixed_version_tests { @@ -782,15 +773,15 @@ mod tests { ) .await; - let sorted_run = SortedRun { - id: 0, - sst_views: vec![ + let sorted_run = SortedRun::new( + 0, + [ SsTableView::identity(sst1_v1), SsTableView::identity(sst2_v2), SsTableView::identity(sst3_v1), SsTableView::identity(sst4_v2), ], - }; + ); // when: iterating over the sorted run let mut iter = SortedRunIterator::new_owned_initialized( @@ -855,15 +846,15 @@ mod tests { ) .await; - let sorted_run = SortedRun { - id: 0, - sst_views: vec![ + let sorted_run = SortedRun::new( + 0, + [ SsTableView::identity(sst1_v1), SsTableView::identity(sst2_v2), SsTableView::identity(sst3_v1), SsTableView::identity(sst4_v2), ], - }; + ); let mut iter = SortedRunIterator::new_owned_initialized( .., diff --git a/slatedb/src/subcompaction.rs b/slatedb/src/subcompaction.rs index 2abaa11a2a..085f900766 100644 --- a/slatedb/src/subcompaction.rs +++ b/slatedb/src/subcompaction.rs @@ -121,7 +121,7 @@ pub(crate) async fn plan_subcompaction_ranges( .chain( sorted_runs .iter() - .flat_map(|sr| sr.sst_views.iter().cloned()), + .flat_map(|sr| sr.sst_views().iter().cloned()), ) .collect(); diff --git a/slatedb/src/test_utils.rs b/slatedb/src/test_utils.rs index 6eacf16021..4d3ebb6daa 100644 --- a/slatedb/src/test_utils.rs +++ b/slatedb/src/test_utils.rs @@ -466,10 +466,7 @@ pub(crate) async fn build_sorted_runs( let ssts = write_ssts(table_store, entries, max_sst_size).await; sr_ssts.extend(ssts.into_iter().map(SsTableView::identity)); } - sorted_runs.push(SortedRun { - id: sr_id as u32, - sst_views: sr_ssts, - }); + sorted_runs.push(SortedRun::new(sr_id as u32, sr_ssts)); } sorted_runs diff --git a/slatedb/src/utils.rs b/slatedb/src/utils.rs index d893382504..e7d16ef1b2 100644 --- a/slatedb/src/utils.rs +++ b/slatedb/src/utils.rs @@ -390,7 +390,7 @@ pub(crate) fn sign_extend(val: u32, bits: u8) -> i32 { /// Returns: /// - The effective max parallelism. pub(crate) fn compute_max_parallel(l0_count: usize, srs: &[SortedRun], cap: usize) -> usize { - let total_ssts = l0_count + srs.iter().map(|sr| sr.sst_views.len()).sum::(); + let total_ssts = l0_count + srs.iter().map(|sr| sr.sst_views().len()).sum::(); total_ssts.min(cap).max(1) } @@ -416,7 +416,7 @@ pub(crate) fn estimate_bytes_before_key(sorted_runs: &[SortedRun], key: &Bytes) return 0; }; sorted_run - .sst_views + .sst_views() .iter() .take(idx) .map(|sst| sst.estimate_size()) @@ -1328,19 +1328,19 @@ mod tests { #[test] fn test_estimate_bytes_before_key() { - let run1 = SortedRun { - id: 1, - sst_views: vec![ + let run1 = SortedRun::new( + 1, + [ make_sst_view("a", 10), make_sst_view("k", 20), // k < m < z, so only "a" counts make_sst_view("z", 30), ], - }; - let run2 = SortedRun { - id: 2, + ); + let run2 = SortedRun::new( + 2, // f < m < ..., so only "b" counts - sst_views: vec![make_sst_view("b", 40), make_sst_view("f", 50)], - }; + [make_sst_view("b", 40), make_sst_view("f", 50)], + ); let key = Bytes::from("m"); let total = estimate_bytes_before_key(&[run1, run2], &key); From 8cab540d14bc88d6808c6d96bdaf6218c2925508 Mon Sep 17 00:00:00 2001 From: Chris Date: Thu, 30 Jul 2026 17:02:50 -0700 Subject: [PATCH 11/65] Disable descending scan checks in DST (#1996) --- slatedb-dst/src/utils.rs | 8 +++----- 1 file changed, 3 insertions(+), 5 deletions(-) diff --git a/slatedb-dst/src/utils.rs b/slatedb-dst/src/utils.rs index 3bb05f3f41..0bbd67b43a 100644 --- a/slatedb-dst/src/utils.rs +++ b/slatedb-dst/src/utils.rs @@ -93,11 +93,9 @@ pub fn build_reader_options(rand: &DbRand) -> DbReaderOptions { pub fn build_scan_options(rand: &DbRand, read_durability: DurabilityLevel) -> ScanOptions { let mut rng = rand.rng(); let read_ahead_options = [1, 4 * 1024, 64 * 1024, MIB_1]; - let order = if rng.random_bool(0.5) { - IterationOrder::Ascending - } else { - IterationOrder::Descending - }; + // Descending sorted-run iteration is currently broken. + // See https://github.com/slatedb/slatedb/pull/1995. + let order = IterationOrder::Ascending; ScanOptions::new() .with_durability_filter(read_durability) From b25a5b8abead88cf339ec0e6fa449abea8057ea5 Mon Sep 17 00:00:00 2001 From: Chris Date: Thu, 30 Jul 2026 17:29:55 -0700 Subject: [PATCH 12/65] Deflake object store cache metrics test (#1997) --- slatedb/src/db.rs | 90 ++++++++++++++++++++++++++++++++++++----------- 1 file changed, 70 insertions(+), 20 deletions(-) diff --git a/slatedb/src/db.rs b/slatedb/src/db.rs index 7bb64eed8a..cde06e2f62 100644 --- a/slatedb/src/db.rs +++ b/slatedb/src/db.rs @@ -2299,6 +2299,35 @@ mod tests { }) } + fn lookup_object_store_api_request_count( + recorder: &DefaultMetricsRecorder, + component: &'static str, + store_type: &'static str, + op: &'static str, + api: &'static str, + ) -> i64 { + lookup_metric_with_labels( + recorder, + OBJECT_STORE_REQUEST_COUNT, + &object_store_labels(component, store_type, op, api), + ) + .unwrap_or(0) + } + + fn lookup_object_store_api_histogram_count( + recorder: &DefaultMetricsRecorder, + component: &'static str, + store_type: &'static str, + op: &'static str, + api: &'static str, + ) -> u64 { + lookup_object_store_histogram_count( + recorder, + &object_store_labels(component, store_type, op, api), + ) + .unwrap_or(0) + } + fn lookup_object_store_op_request_count( recorder: &DefaultMetricsRecorder, component: &'static str, @@ -2319,12 +2348,7 @@ mod tests { apis.iter() .map(|api| { - lookup_metric_with_labels( - recorder, - OBJECT_STORE_REQUEST_COUNT, - &object_store_labels(component, store_type, op, api), - ) - .unwrap_or(0) + lookup_object_store_api_request_count(recorder, component, store_type, op, api) }) .sum() } @@ -2349,11 +2373,7 @@ mod tests { apis.iter() .map(|api| { - lookup_object_store_histogram_count( - recorder, - &object_store_labels(component, store_type, op, api), - ) - .unwrap_or(0) + lookup_object_store_api_histogram_count(recorder, component, store_type, op, api) }) .sum() } @@ -11781,26 +11801,56 @@ mod tests { .await .unwrap(); - let requests_before = - lookup_object_store_op_request_count(&metrics_recorder, "db", "main", "get"); + let requests_before = lookup_object_store_api_request_count( + &metrics_recorder, + "db", + "main", + "get", + "get_range", + ); + let histograms_before = lookup_object_store_api_histogram_count( + &metrics_recorder, + "db", + "main", + "get", + "get_range", + ); let _val = kv_store.get(b"test_key").await.unwrap(); - let requests_after_first = - lookup_object_store_op_request_count(&metrics_recorder, "db", "main", "get"); + let requests_after_first = lookup_object_store_api_request_count( + &metrics_recorder, + "db", + "main", + "get", + "get_range", + ); let got = kv_store.get(b"test_key").await.unwrap(); - let requests_after_second = - lookup_object_store_op_request_count(&metrics_recorder, "db", "main", "get"); + let requests_after_second = lookup_object_store_api_request_count( + &metrics_recorder, + "db", + "main", + "get", + "get_range", + ); + let histograms_after_second = lookup_object_store_api_histogram_count( + &metrics_recorder, + "db", + "main", + "get", + "get_range", + ); // The instrumented store sits above the object store cache and counts // logical read calls whether they are served from the cache or the - // remote store. + // remote store. Restrict the assertion to range reads so background + // manifest polling, which uses plain get calls, cannot affect it. // A point get reads the single-part SST in three sub-ranges (index, // filter and block). assert_eq!(requests_after_first, requests_before + 3); assert_eq!(got, Some(Bytes::from_static(b"test_value"))); assert_eq!(requests_after_second, requests_after_first + 3); assert_eq!( - lookup_object_store_op_histogram_count(&metrics_recorder, "db", "main", "get"), - requests_after_second as u64 + histograms_after_second - histograms_before, + (requests_after_second - requests_before) as u64 ); kv_store.close().await.unwrap(); } From 1bc20085e7c25a6a8d91016cce9ee90f6e77e944 Mon Sep 17 00:00:00 2001 From: Aether <98370028+Aetherance@users.noreply.github.com> Date: Sat, 1 Aug 2026 00:01:20 +0800 Subject: [PATCH 13/65] Allow closing a DB without flushing the active memtable (#1948) --- slatedb/src/config.rs | 26 ++++++++++ slatedb/src/db.rs | 113 ++++++++++++++++++++++++++++++++++++++++-- 2 files changed, 134 insertions(+), 5 deletions(-) diff --git a/slatedb/src/config.rs b/slatedb/src/config.rs index b3bfd74948..19e99a2905 100644 --- a/slatedb/src/config.rs +++ b/slatedb/src/config.rs @@ -461,6 +461,32 @@ impl Default for FlushOptions { } } +/// Options controlling how a database is closed. +#[derive(Clone, Debug)] +pub struct CloseOptions { + /// Whether to trigger a final flush of the active memtable before closing. + /// + /// When `false`, memtables already being flushed continue through the + /// existing shutdown pipeline. Defaults to `true`. + pub flush_memtables: bool, +} + +impl Default for CloseOptions { + fn default() -> Self { + Self { + flush_memtables: true, + } + } +} + +impl CloseOptions { + /// Configure whether the active memtable is flushed before closing. + pub fn with_flush_memtables(mut self, flush_memtables: bool) -> Self { + self.flush_memtables = flush_memtables; + self + } +} + /// Configuration for client write operations. `WriteOptions` is supplied for each /// write call and controls the behavior of the write. #[derive(Clone, Debug, Default)] diff --git a/slatedb/src/db.rs b/slatedb/src/db.rs index cde06e2f62..bb2f9b2a12 100644 --- a/slatedb/src/db.rs +++ b/slatedb/src/db.rs @@ -48,8 +48,8 @@ use crate::bytes_range::{ByteRangeBounds, BytesRange}; use crate::cached_object_store::CachedObjectStore; use crate::clock::MonotonicClock; use crate::config::{ - FlushOptions, FlushType, MergeOptions, PutOptions, ReadOptions, ScanOptions, Settings, - WriteOptions, + CloseOptions, FlushOptions, FlushType, MergeOptions, PutOptions, ReadOptions, ScanOptions, + Settings, WriteOptions, }; use crate::db_common::extract_segment_prefix; use crate::db_iter::{DbIterator, DbRecencyIterator}; @@ -681,6 +681,15 @@ impl Db { /// } /// ``` pub async fn close(&self) -> Result<(), crate::Error> { + self.close_with_options(CloseOptions::default()).await + } + + /// Close the database with custom options. + /// + /// Setting [`CloseOptions::flush_memtables`] to `false` skips the final + /// active memtable flush. Memtables already being flushed are allowed to + /// finish, and writes that are not durable may be lost. + pub async fn close_with_options(&self, options: CloseOptions) -> Result<(), crate::Error> { let should_flush = match self.status().close_reason { // If already closed, don't close again. Some(CloseReason::Clean) => return Err(SlateDBError::Closed.into()), @@ -689,8 +698,9 @@ impl Db { // run when in a failed state (vs. a clean closure, which will return // Error::Closed(CloseReason::Clean) on subsequent calls). Some(_) => false, - // Flush outstanding writes if the database is still open. - None => true, + // Flush outstanding writes if the database is still open and the + // caller requested a final memtable flush. + None => options.flush_memtables, }; // Mark the database as closed before flushing. @@ -2222,7 +2232,7 @@ mod tests { use crate::config::DurabilityLevel::{Memory, Remote}; use crate::config::MetricLevel; use crate::config::{ - CheckpointOptions, CompactionWorkerOptions, CompactorOptions, + CheckpointOptions, CloseOptions, CompactionWorkerOptions, CompactorOptions, GarbageCollectorDirectoryOptions, GarbageCollectorOptions, ObjectStoreCacheOptions, PutOptions, ScanOptions, Settings, SstBlockSize, Ttl, WriteOptions, }; @@ -3468,6 +3478,99 @@ mod tests { ); } + #[tokio::test] + async fn test_close_with_options_default_flushes_final_memtable() { + let object_store: Arc = Arc::new(InMemory::new()); + let mut settings = test_db_options(0, 1024, None); + settings.flush_interval = None; + let metrics_recorder = Arc::new(DefaultMetricsRecorder::new()); + let db = Db::builder( + "/tmp/test_close_with_options_default_flushes_final_memtable", + object_store, + ) + .with_settings(settings) + .with_metrics_recorder(metrics_recorder.clone()) + .build() + .await + .unwrap(); + + db.put(b"test_key", b"test_value").await.unwrap(); + + assert_eq!( + lookup_metric(&metrics_recorder, crate::db_stats::L0_FLUSH_BYTES).unwrap_or(0), + 0 + ); + + db.close_with_options(CloseOptions::default()) + .await + .unwrap(); + + assert!( + lookup_metric(&metrics_recorder, crate::db_stats::L0_FLUSH_BYTES).unwrap() > 0, + "expected L0 flush during close with default options" + ); + } + + #[tokio::test] + async fn test_close_with_options_skips_final_flush() { + let object_store: Arc = Arc::new(InMemory::new()); + let mut settings = test_db_options(0, 1024, None); + settings.flush_interval = None; + let metrics_recorder = Arc::new(DefaultMetricsRecorder::new()); + let db = Db::builder( + "/tmp/test_close_with_options_skips_final_flush", + object_store, + ) + .with_settings(settings) + .with_metrics_recorder(metrics_recorder.clone()) + .build() + .await + .unwrap(); + + db.put(b"test_key", b"test_value").await.unwrap(); + + db.close_with_options(CloseOptions::default().with_flush_memtables(false)) + .await + .unwrap(); + + assert_eq!( + lookup_metric(&metrics_recorder, crate::db_stats::L0_FLUSH_BYTES).unwrap_or(0), + 0 + ); + } + + #[cfg(feature = "wal_disable")] + #[tokio::test] + async fn test_close_without_flush_fails_pending_durability_wait() { + let object_store: Arc = Arc::new(InMemory::new()); + let mut settings = test_db_options(0, 1024, None); + settings.flush_interval = None; + settings.wal_enabled = false; + let db = Db::builder( + "/tmp/test_close_without_flush_fails_pending_durability_wait", + object_store, + ) + .with_settings(settings) + .build() + .await + .unwrap(); + + let handle = db.put(b"key", b"value").await.unwrap(); + + db.close_with_options(CloseOptions::default().with_flush_memtables(false)) + .await + .unwrap(); + + let error = tokio::time::timeout(Duration::from_secs(5), handle.await_durable()) + .await + .expect("durability wait remained blocked after close") + .expect_err("discarded write should not become durable"); + assert!(matches!( + error.kind(), + crate::ErrorKind::Closed(CloseReason::Clean) + )); + } + #[tokio::test] async fn test_memtable_write_bytes_matches_batch_payload() { let object_store: Arc = Arc::new(InMemory::new()); From 03b18f6a6c719baf310fcbe0d7267deac2bed580 Mon Sep 17 00:00:00 2001 From: Chris Date: Fri, 31 Jul 2026 10:10:27 -0700 Subject: [PATCH 14/65] site: add Prisma to adopter listings (#1999) --- README.md | 1 + website/public/img/logos/prisma.svg | 3 +++ website/src/pages/index.astro | 14 ++++++++++++++ 3 files changed, 18 insertions(+) create mode 100644 website/public/img/logos/prisma.svg diff --git a/README.md b/README.md index 86fb25f8f1..63870960d0 100644 --- a/README.md +++ b/README.md @@ -161,6 +161,7 @@ See who's using SlateDB. - [Massive](https://massive.com) - [Merklemap](https://merklemap.com) - [OpenData](https://www.opendata.dev) +- [Prisma](https://www.prisma.io) - [Responsive](https://responsive.dev) - [s2-lite](https://github.com/s2-streamstore/s2) - [SQLync](https://sqlync.com) diff --git a/website/public/img/logos/prisma.svg b/website/public/img/logos/prisma.svg new file mode 100644 index 0000000000..0f42a21023 --- /dev/null +++ b/website/public/img/logos/prisma.svg @@ -0,0 +1,3 @@ + + + diff --git a/website/src/pages/index.astro b/website/src/pages/index.astro index 02788e0862..2cd2b328ba 100644 --- a/website/src/pages/index.astro +++ b/website/src/pages/index.astro @@ -104,6 +104,7 @@ const jsonLd = { // The "used by" companies — even row, all logos rendered at the same height. const usedBy = [ { name: 'Dropbox', href: 'https://www.dropbox.com/', logo: '/img/logos/dropbox.svg', scale: 1.2 }, + { name: 'Prisma', href: 'https://www.prisma.io/', logo: '/img/logos/prisma.svg' }, { name: 'HelixDB', href: 'https://www.helix-db.com/', logo: '/img/logos/helixdb.png', scale: 0.8 }, { name: 'TensorLake', href: 'https://www.tensorlake.ai/', logo: '/img/logos/tensorlake.svg', scale: 1.2 }, { name: 'S2', href: 'https://s2.dev/', logo: '/img/logos/s2.svg' }, @@ -925,6 +926,19 @@ const osLinks = [ filter: grayscale(0%); } + @media (max-width: 640px) { + .usedby-row { + display: flex; + flex-wrap: wrap; + justify-content: center; + column-gap: 0.75rem; + } + .usedby-logo { + flex: 0 0 calc((100% - 1.5rem) / 3); + width: calc((100% - 1.5rem) / 3); + } + } + /* ---------- Editorial prose ---------- */ .prose { From 90117dc27a567a1a985145d94be6b0522386eb09 Mon Sep 17 00:00:00 2001 From: Chris Date: Fri, 31 Jul 2026 11:35:58 -0700 Subject: [PATCH 15/65] Add `consume_budget` in hot loops to help Tokio (#1998) --- slatedb/src/batch.rs | 2 ++ slatedb/src/compactor_executor.rs | 3 +++ slatedb/src/db_iter.rs | 10 ++++++++-- slatedb/src/filter_iterator.rs | 2 ++ slatedb/src/flush.rs | 4 ++++ slatedb/src/format/sst.rs | 20 +++++++++++++++----- slatedb/src/mem_table.rs | 2 ++ slatedb/src/merge_operator.rs | 2 ++ slatedb/src/sst_builder.rs | 4 ---- 9 files changed, 38 insertions(+), 11 deletions(-) diff --git a/slatedb/src/batch.rs b/slatedb/src/batch.rs index 43b75262d2..9a7244635f 100644 --- a/slatedb/src/batch.rs +++ b/slatedb/src/batch.rs @@ -470,6 +470,8 @@ impl RowEntryIterator for WriteBatchIterator { IterationOrder::Descending => entry.key.as_ref() > next_key, } { self.iter.next(); + // Keep in-memory seeking cooperative. + tokio::task::coop::consume_budget().await; } else { break; } diff --git a/slatedb/src/compactor_executor.rs b/slatedb/src/compactor_executor.rs index 8292f6e8ec..9fff4281e7 100644 --- a/slatedb/src/compactor_executor.rs +++ b/slatedb/src/compactor_executor.rs @@ -858,6 +858,9 @@ impl TokioCompactionExecutorInner { let total_bytes = start_bytes_processed + all_iter.bytes_processed(); progress(total_bytes, &output_ssts); } + + // Keep cached compaction work cooperative. + tokio::task::coop::consume_budget().await; } // Drain the in-flight close, then flush the final partial SST. Order diff --git a/slatedb/src/db_iter.rs b/slatedb/src/db_iter.rs index 2886b3e65c..5246d84b42 100644 --- a/slatedb/src/db_iter.rs +++ b/slatedb/src/db_iter.rs @@ -284,7 +284,10 @@ impl DbIterator { Err(error) } else { let result = loop { - match self.iter.next().await { + let next = self.iter.next().await; + // Keep cached iteration cooperative. + tokio::task::coop::consume_budget().await; + match next { Ok(Some(entry)) => match entry.value { ValueDeletable::Tombstone => continue, _ => break Ok(Some(entry)), @@ -422,7 +425,10 @@ impl DbRecencyIterator { self.current_initialized = true; } - match iter.next().await { + let next = iter.next().await; + // Keep cached iteration cooperative. + tokio::task::coop::consume_budget().await; + match next { Ok(Some(entry)) => return Ok(Some(entry)), Ok(None) => { self.iters.pop_front(); diff --git a/slatedb/src/filter_iterator.rs b/slatedb/src/filter_iterator.rs index 2ba53508b8..c6ce9f20fc 100644 --- a/slatedb/src/filter_iterator.rs +++ b/slatedb/src/filter_iterator.rs @@ -41,6 +41,8 @@ impl RowEntryIterator for FilterIterator { if (self.predicate)(&entry) { return Ok(Some(entry)); } + // Keep filtered scans cooperative. + tokio::task::coop::consume_budget().await; } Ok(None) } diff --git a/slatedb/src/flush.rs b/slatedb/src/flush.rs index 2563f2a608..bdd5e1e135 100644 --- a/slatedb/src/flush.rs +++ b/slatedb/src/flush.rs @@ -37,6 +37,8 @@ impl DbInner { while let Some(entry) = iter.next().await? { sst_builder.add(entry).await?; any = true; + // Keep cached flush work cooperative. + tokio::task::coop::consume_budget().await; } if !any { return Ok(None); @@ -131,6 +133,8 @@ impl DbInner { } current_builder.add(entry).await?; current_has_entry = true; + // Keep cached flush work cooperative. + tokio::task::coop::consume_budget().await; } if current_has_entry { out.push(EncodedSegmentSst { diff --git a/slatedb/src/format/sst.rs b/slatedb/src/format/sst.rs index 069f34dfc8..7a34982efd 100644 --- a/slatedb/src/format/sst.rs +++ b/slatedb/src/format/sst.rs @@ -536,7 +536,12 @@ pub(crate) async fn compress_and_transform( ) -> Result { let compressed = match compression_codec { None => data, - Some(c) => compress(data, c)?, + Some(c) => { + let compressed = compress(data, c)?; + // Account for CPU-only compression work. + tokio::task::coop::consume_budget().await; + compressed + } }; let transformed = transform(compressed, block_transformer).await?; let checksum = crc32fast::hash(&transformed); @@ -600,10 +605,15 @@ pub(crate) async fn transform( block_transformer: Option<&Arc>, ) -> Result { let transformed = match block_transformer { - Some(t) => t - .encode(data) - .await - .map_err(|_| SlateDBError::BlockTransformError)?, + Some(t) => { + let transformed = t + .encode(data) + .await + .map_err(|_| SlateDBError::BlockTransformError)?; + // Account for CPU-only transformation work. + tokio::task::coop::consume_budget().await; + transformed + } None => data, }; Ok(transformed) diff --git a/slatedb/src/mem_table.rs b/slatedb/src/mem_table.rs index 23cce96a1d..21155b2f1c 100644 --- a/slatedb/src/mem_table.rs +++ b/slatedb/src/mem_table.rs @@ -212,6 +212,8 @@ impl RowEntryIterator for MemTableIterator { let front = self.borrow_item().clone(); if front.is_some_and(|record| record.key < next_key) { self.next_sync(); + // Keep in-memory seeking cooperative. + tokio::task::coop::consume_budget().await; } else { return Ok(()); } diff --git a/slatedb/src/merge_operator.rs b/slatedb/src/merge_operator.rs index b1ee07fde6..6e21f78e29 100644 --- a/slatedb/src/merge_operator.rs +++ b/slatedb/src/merge_operator.rs @@ -403,6 +403,8 @@ impl MergeOperatorIterator { } next = self.delegate.next().await?; + // Keep cached operand scans cooperative. + tokio::task::coop::consume_budget().await; } else { break None; } diff --git a/slatedb/src/sst_builder.rs b/slatedb/src/sst_builder.rs index b26bfbca43..edc5853e8b 100644 --- a/slatedb/src/sst_builder.rs +++ b/slatedb/src/sst_builder.rs @@ -322,10 +322,6 @@ impl EncodedSsTableBuilder { self.blocks.push_back(block); self.first_key = None; - // Block encoding (compression/block transformer) is CPU-heavy. - // Give runtime a chance to run other tasks after each block. - tokio::task::yield_now().await; - Ok(Some(block_size)) } From e67d7d492d4bb5ae7a44b527abab8408af1e01cd Mon Sep 17 00:00:00 2001 From: Rohan Date: Mon, 3 Aug 2026 18:49:44 -0400 Subject: [PATCH 16/65] [rfc-30 3/N]: add trait and native implementation for wal gc (#2000) --- slatedb/src/garbage_collector.rs | 20 +- slatedb/src/garbage_collector/wal_gc.rs | 285 ++++++++-------- slatedb/src/manifest/mod.rs | 4 - slatedb/src/wal/gc.rs | 415 ++++++++++++++++++++++++ slatedb/src/wal/mod.rs | 14 + 5 files changed, 588 insertions(+), 150 deletions(-) create mode 100644 slatedb/src/wal/gc.rs diff --git a/slatedb/src/garbage_collector.rs b/slatedb/src/garbage_collector.rs index 3098d9b0fa..a4fa093242 100644 --- a/slatedb/src/garbage_collector.rs +++ b/slatedb/src/garbage_collector.rs @@ -24,6 +24,7 @@ use crate::manifest::store::{ManifestStore, StoredManifest}; use crate::manifest::Manifest; use crate::tablestore::TableStore; use crate::utils::WatchableOnceCell; +use crate::wal::gc::{SlateDbWalGc, WalGcMode}; use async_trait::async_trait; use chrono::{DateTime, Utc}; use compacted_gc::CompactedGcTask; @@ -40,7 +41,7 @@ use std::sync::Arc; use std::time::Duration; use tokio::runtime::Handle; use tracing::instrument; -use wal_gc::{WalGcMode, WalGcTask}; +use wal_gc::WalGcTask; mod compacted_gc; mod compactions_gc; @@ -50,6 +51,7 @@ mod manifest_gc; pub mod stats; mod wal_gc; +pub(crate) use filter::retain_allowed_by_gc_filter; pub use filter::GcFilter; pub(crate) const DEFAULT_MIN_AGE: Duration = Duration::from_secs(300); @@ -243,24 +245,30 @@ impl GarbageCollector { system_clock.clone(), )); let wal_gc_task = options.wal_options.map(|wal_options| { - WalGcTask::new( - manifest_store.clone(), + let wal_gc = Arc::new(SlateDbWalGc::new( table_store.clone(), stats.clone(), wal_options, WalGcMode::Regular, gc_filter.clone(), + system_clock.clone(), + )); + WalGcTask::new( + manifest_store.clone(), + wal_gc, + WalGcMode::Regular.resource(), ) }); let wal_fence_gc_task = options.wal_fence_options.map(|wal_fence_options| { - WalGcTask::new( - manifest_store.clone(), + let wal_gc = Arc::new(SlateDbWalGc::new( table_store.clone(), stats.clone(), wal_fence_options, WalGcMode::Fence, gc_filter.clone(), - ) + system_clock.clone(), + )); + WalGcTask::new(manifest_store.clone(), wal_gc, WalGcMode::Fence.resource()) }); let compacted_gc_task = options.compacted_options.map(|compacted_options| { CompactedGcTask::new( diff --git a/slatedb/src/garbage_collector/wal_gc.rs b/slatedb/src/garbage_collector/wal_gc.rs index f4b06c2f54..2ef3be417d 100644 --- a/slatedb/src/garbage_collector/wal_gc.rs +++ b/slatedb/src/garbage_collector/wal_gc.rs @@ -1,51 +1,27 @@ use crate::manifest::Manifest; use crate::{ - config::GarbageCollectorDirectoryOptions, db_state::SsTableId, error::SlateDBError, - manifest::store::ManifestStore, tablestore::TableStore, + error::SlateDBError, + manifest::store::ManifestStore, + wal::{WalFileRange, WalGC}, }; use chrono::{DateTime, Utc}; -use futures::StreamExt; -use log::error; use std::collections::BTreeMap; +use std::ops::Bound; use std::sync::Arc; -use super::filter::retain_allowed_by_gc_filter; -use super::{GcFilter, GcStats, GcTask, GC_DELETE_CONCURRENCY}; -use slatedb_common::object_metadata::IdentifiedObjectMetadata; - -/// Selects which class of WAL object a [`WalGcTask`] collects. -/// -/// Regular WAL SSTs and zero-byte WAL fence objects share the same WAL -/// directory and `SsTableId::Wal` identifier space, but they have separate -/// retention policies. This mode keeps a single task implementation while -/// allowing regular WAL GC and fence WAL GC to run on independent schedules. -#[derive(Debug, Clone, Copy)] -pub(super) enum WalGcMode { - /// Collect non-empty WAL SSTs that are older than the compacted WAL - /// boundary, old enough for retention, and unreferenced by active - /// checkpoint manifests. - Regular, - - /// Collect zero-byte WAL fence objects under the same safety checks as - /// regular WAL GC. - Fence, -} +use super::GcTask; #[derive(Clone)] pub(crate) struct WalGcTask { manifest_store: Arc, - table_store: Arc, - stats: Arc, - wal_options: GarbageCollectorDirectoryOptions, - mode: WalGcMode, - gc_filter: Option>, + wal_gc: Arc, + resource: &'static str, } impl std::fmt::Debug for WalGcTask { fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { f.debug_struct("WalGcTask") - .field("wal_options", &self.wal_options) - .field("mode", &self.mode) + .field("resource", &self.resource.to_string()) .finish() } } @@ -53,136 +29,165 @@ impl std::fmt::Debug for WalGcTask { impl WalGcTask { pub(super) fn new( manifest_store: Arc, - table_store: Arc, - stats: Arc, - wal_options: GarbageCollectorDirectoryOptions, - mode: WalGcMode, - gc_filter: Option>, + wal_gc: Arc, + resource: &'static str, ) -> Self { - WalGcTask { + Self { manifest_store, - table_store, - stats, - wal_options, - mode, - gc_filter, + wal_gc, + resource, } } - fn is_wal_sst_eligible_for_deletion( - utc_now: &DateTime, - wal_sst: &IdentifiedObjectMetadata, - min_age: &chrono::Duration, + fn referenced_wal_ranges( + latest_manifest_id: u64, active_manifests: &BTreeMap, - ) -> bool { - if utc_now.signed_duration_since(wal_sst.metadata.last_modified) <= *min_age { - return false; - } - - let wal_sst_id = wal_sst.id.unwrap_wal_id(); - !active_manifests - .values() - .any(|manifest| manifest.has_wal_sst_reference(wal_sst_id)) - } - - fn wal_sst_min_age(&self) -> chrono::Duration { - chrono::Duration::from_std(self.wal_options.min_age).expect("invalid duration") - } - - /// Deletes the given WAL SSTs from the table store. - /// - /// In case of dryrun, the actual deletion doesn't happen. - async fn maybe_delete_wal_ssts(&self, sst_ids: Vec) { - if self.wal_options.dry_run { - if !sst_ids.is_empty() { - log::info!( - "dry run: skipping {} deletion [count={}]", - self.resource(), - sst_ids.len() - ); - if matches!(self.mode, WalGcMode::Fence) { - log::info!( - "WAL fence GC is dry-run by default. This is a conservative setting. \ - Set wal_fence_options.dry_run=false and use a conservative min_age to enable. \ - Silence this log with wal_fence_options=None. See #352 for details." - ); - } - } - for id in sst_ids { - log::debug!( - "dry run: would delete {} but skipped [id={:?}]", - self.resource(), - id - ); - } - return; - } - - futures::stream::iter(sst_ids) - .for_each_concurrent(GC_DELETE_CONCURRENCY, |id| async move { - if let Err(e) = self.table_store.delete_sst(&id).await { - error!("error deleting WAL SST [id={:?}, error={}]", id, e); + ) -> Vec { + active_manifests + .iter() + .map(|(manifest_id, manifest)| { + if *manifest_id == latest_manifest_id { + // Keep the current compaction boundary and everything after it. Retaining the + // boundary matches the existing GC protocol and protects concurrent writers. + WalFileRange( + Bound::Included(manifest.core.replay_after_wal_id), + Bound::Unbounded, + ) } else { - match self.mode { - WalGcMode::Regular => self.stats.gc_wal_count.increment(1), - WalGcMode::Fence => self.stats.gc_wal_fence_count.increment(1), - } + // A checkpoint only references WALs that must be replayed for its manifest. + WalFileRange( + Bound::Excluded(manifest.core.replay_after_wal_id), + Bound::Excluded(manifest.core.next_wal_sst_id), + ) } }) - .await; + .collect() } } impl GcTask for WalGcTask { - /// Collect garbage from the WAL SSTs. This will delete any WAL SSTs that meet - /// the following conditions: - /// - not referenced by an active checkpoint - /// - older than the minimum age specified in the options - /// - older than the last compacted WAL SST. - async fn collect(&self, utc_now: DateTime) -> Result<(), SlateDBError> { + /// Resolve the WAL ranges referenced by the current manifest and active checkpoints, then + /// delegate collection to the configured WAL implementation. + async fn collect(&self, _utc_now: DateTime) -> Result<(), SlateDBError> { let latest_manifest = self.manifest_store.read_latest_manifest().await?; let active_manifests = self .manifest_store .read_referenced_manifests(latest_manifest.id, &latest_manifest.manifest) .await?; - let last_compacted_wal_sst_id = latest_manifest.manifest.core.replay_after_wal_id; - let min_age = self.wal_sst_min_age(); - let ssts_to_delete = self - .table_store - .list_wal_ssts(..last_compacted_wal_sst_id) - .await? - .into_iter() - .filter(|wal_sst| match self.mode { - // In regular mode, only consider WAL SSTs with size > 0 for deletion. - WalGcMode::Regular => wal_sst.metadata.size > 0, - // In fence mode, only consider zero-byte WAL SSTs for deletion. - WalGcMode::Fence => wal_sst.metadata.size == 0, - }) - // Respect min_age and any WAL references held by active checkpoint manifests. - .filter(|wal_sst| { - Self::is_wal_sst_eligible_for_deletion( - &utc_now, - wal_sst, - &min_age, - &active_manifests, - ) - }) - .collect::>(); - let ssts_to_delete = retain_allowed_by_gc_filter(&self.gc_filter, ssts_to_delete).await; - let sst_ids_to_delete = ssts_to_delete - .into_iter() - .map(|wal_sst| wal_sst.id) - .collect::>(); - - self.maybe_delete_wal_ssts(sst_ids_to_delete).await; + let referenced_ranges = Self::referenced_wal_ranges(latest_manifest.id, &active_manifests); - Ok(()) + self.wal_gc + .collect(referenced_ranges) + .await + .map_err(Into::into) } fn resource(&self) -> &str { - match self.mode { - WalGcMode::Regular => "WAL", - WalGcMode::Fence => "WAL fence", + self.resource + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::checkpoint::Checkpoint; + use crate::manifest::store::StoredManifest; + use crate::manifest::ManifestCore; + use crate::wal::WalError; + use async_trait::async_trait; + use object_store::memory::InMemory; + use object_store::path::Path; + use object_store::ObjectStore; + use slatedb_common::clock::DefaultSystemClock; + use std::sync::Mutex; + use uuid::Uuid; + + #[derive(Default)] + struct RecordingWalGc { + calls: Mutex>>, + } + + impl RecordingWalGc { + fn calls(&self) -> Vec> { + self.calls.lock().unwrap().clone() + } + } + + #[async_trait] + impl WalGC for RecordingWalGc { + async fn collect(&self, referenced_ranges: Vec) -> Result<(), WalError> { + self.calls.lock().unwrap().push(referenced_ranges); + Ok(()) } } + + #[test] + fn test_referenced_wal_ranges() { + let mut checkpoint_core = ManifestCore::new(); + checkpoint_core.replay_after_wal_id = 2; + checkpoint_core.next_wal_sst_id = 6; + + let mut current_core = ManifestCore::new(); + current_core.replay_after_wal_id = 5; + current_core.next_wal_sst_id = 8; + + let active_manifests = BTreeMap::from([ + (1, Manifest::initial(checkpoint_core)), + (2, Manifest::initial(current_core)), + ]); + + assert_eq!( + WalGcTask::referenced_wal_ranges(2, &active_manifests), + vec![ + WalFileRange(Bound::Excluded(2), Bound::Excluded(6)), + WalFileRange(Bound::Included(5), Bound::Unbounded), + ] + ); + } + + #[tokio::test] + async fn test_collect_calls_wal_gc_with_referenced_ranges() { + let object_store: Arc = Arc::new(InMemory::new()); + let manifest_store = Arc::new(ManifestStore::new( + &Path::from("/test/wal-gc-ranges"), + object_store, + )); + + let mut checkpoint_core = ManifestCore::new(); + checkpoint_core.replay_after_wal_id = 2; + checkpoint_core.next_wal_sst_id = 6; + let mut stored_manifest = StoredManifest::create_new_db( + manifest_store.clone(), + checkpoint_core, + Arc::new(DefaultSystemClock::new()), + ) + .await + .unwrap(); + let checkpoint_manifest_id = stored_manifest.id(); + + let mut dirty = stored_manifest.prepare_dirty().unwrap(); + dirty.value.core.replay_after_wal_id = 5; + dirty.value.core.next_wal_sst_id = 8; + dirty.value.core.checkpoints.push(Checkpoint { + id: Uuid::new_v4(), + manifest_id: checkpoint_manifest_id, + expire_time: None, + create_time: Utc::now(), + name: None, + }); + stored_manifest.update(dirty).await.unwrap(); + + let wal_gc = Arc::new(RecordingWalGc::default()); + let task = WalGcTask::new(manifest_store, wal_gc.clone(), "WAL"); + + task.collect(Utc::now()).await.unwrap(); + + assert_eq!( + wal_gc.calls(), + vec![vec![ + WalFileRange(Bound::Excluded(2), Bound::Excluded(6)), + WalFileRange(Bound::Included(5), Bound::Unbounded), + ]] + ); + } } diff --git a/slatedb/src/manifest/mod.rs b/slatedb/src/manifest/mod.rs index 5091e23b59..bd4644f6b1 100644 --- a/slatedb/src/manifest/mod.rs +++ b/slatedb/src/manifest/mod.rs @@ -1543,10 +1543,6 @@ impl Manifest { .collect() } - pub(crate) fn has_wal_sst_reference(&self, wal_sst_id: u64) -> bool { - wal_sst_id > self.core.replay_after_wal_id && wal_sst_id < self.core.next_wal_sst_id - } - /// Shrinks each `ExternalDb.sst_ids` to only IDs still referenced by this manifest's /// L0 and compacted sorted runs. `ExternalDb` entries are retained even when their /// `sst_ids` becomes empty — detaching a clone from its parent is done by the GC, diff --git a/slatedb/src/wal/gc.rs b/slatedb/src/wal/gc.rs new file mode 100644 index 0000000000..e583b11b2c --- /dev/null +++ b/slatedb/src/wal/gc.rs @@ -0,0 +1,415 @@ +use crate::config::GarbageCollectorDirectoryOptions; +use crate::db_state::SsTableId; +use crate::garbage_collector::stats::GcStats; +use crate::garbage_collector::{retain_allowed_by_gc_filter, GcFilter, GC_DELETE_CONCURRENCY}; +use crate::tablestore::TableStore; +use crate::wal::{WalError, WalFileRange, WalGC}; +use async_trait::async_trait; +use chrono::{DateTime, Utc}; +use futures::StreamExt; +use log::error; +use slatedb_common::clock::SystemClock; +use slatedb_common::object_metadata::IdentifiedObjectMetadata; +use std::ops::Bound; +use std::sync::Arc; + +/// Selects which class of SlateDB WAL object is collected. +/// +/// Regular WAL SSTs and zero-byte WAL fence objects share the same WAL +/// directory and `SsTableId::Wal` identifier space, but they have separate +/// retention policies. This mode keeps a single implementation while +/// allowing regular WAL GC and fence WAL GC to run on independent schedules. +#[derive(Debug, Clone, Copy)] +pub(crate) enum WalGcMode { + /// Collect non-empty WAL SSTs that are old enough for retention and unreferenced by active + /// manifests. + Regular, + + /// Collect zero-byte WAL fence objects under the same safety checks as regular WAL GC. + Fence, +} + +impl WalGcMode { + pub(crate) fn resource(self) -> &'static str { + match self { + WalGcMode::Regular => "WAL", + WalGcMode::Fence => "WAL fence", + } + } +} + +#[derive(Clone)] +pub(crate) struct SlateDbWalGc { + table_store: Arc, + stats: Arc, + wal_options: GarbageCollectorDirectoryOptions, + mode: WalGcMode, + gc_filter: Option>, + system_clock: Arc, +} + +impl std::fmt::Debug for SlateDbWalGc { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("SlateDbWalGc") + .field("wal_options", &self.wal_options) + .field("mode", &self.mode) + .finish() + } +} + +impl SlateDbWalGc { + pub(crate) fn new( + table_store: Arc, + stats: Arc, + wal_options: GarbageCollectorDirectoryOptions, + mode: WalGcMode, + gc_filter: Option>, + system_clock: Arc, + ) -> Self { + Self { + table_store, + stats, + wal_options, + mode, + gc_filter, + system_clock, + } + } + + fn is_wal_sst_eligible_for_deletion( + utc_now: &DateTime, + wal_sst: &IdentifiedObjectMetadata, + min_age: &chrono::Duration, + referenced_ranges: &[WalFileRange], + ) -> bool { + if utc_now.signed_duration_since(wal_sst.metadata.last_modified) <= *min_age { + return false; + } + + let wal_sst_id = wal_sst.id.unwrap_wal_id(); + !referenced_ranges + .iter() + .any(|range| Self::range_contains(range, wal_sst_id)) + } + + fn range_contains(range: &WalFileRange, wal_sst_id: u64) -> bool { + let after_start = match &range.0 { + Bound::Included(start) => wal_sst_id >= *start, + Bound::Excluded(start) => wal_sst_id > *start, + Bound::Unbounded => true, + }; + let before_end = match &range.1 { + Bound::Included(end) => wal_sst_id <= *end, + Bound::Excluded(end) => wal_sst_id < *end, + Bound::Unbounded => true, + }; + after_start && before_end + } + + fn wal_sst_min_age(&self) -> chrono::Duration { + chrono::Duration::from_std(self.wal_options.min_age).expect("invalid duration") + } + + /// Deletes the given WAL SSTs from the table store. + /// + /// In case of dryrun, the actual deletion doesn't happen. + async fn maybe_delete_wal_ssts(&self, sst_ids: Vec) { + if self.wal_options.dry_run { + if !sst_ids.is_empty() { + log::info!( + "dry run: skipping {} deletion [count={}]", + self.mode.resource(), + sst_ids.len() + ); + if matches!(self.mode, WalGcMode::Fence) { + log::info!( + "WAL fence GC is dry-run by default. This is a conservative setting. \ + Set wal_fence_options.dry_run=false and use a conservative min_age to enable. \ + Silence this log with wal_fence_options=None. See #352 for details." + ); + } + } + for id in sst_ids { + log::debug!( + "dry run: would delete {} but skipped [id={:?}]", + self.mode.resource(), + id + ); + } + return; + } + + futures::stream::iter(sst_ids) + .for_each_concurrent(GC_DELETE_CONCURRENCY, |id| async move { + if let Err(e) = self.table_store.delete_sst(&id).await { + error!("error deleting WAL SST [id={:?}, error={}]", id, e); + } else { + match self.mode { + WalGcMode::Regular => self.stats.gc_wal_count.increment(1), + WalGcMode::Fence => self.stats.gc_wal_fence_count.increment(1), + } + } + }) + .await; + } +} + +#[async_trait] +impl WalGC for SlateDbWalGc { + async fn collect(&self, referenced_ranges: Vec) -> Result<(), WalError> { + let utc_now = self.system_clock.now(); + let min_age = self.wal_sst_min_age(); + let ssts_to_delete = self + .table_store + .list_wal_ssts(..) + .await? + .into_iter() + .filter(|wal_sst| match self.mode { + // In regular mode, only consider WAL SSTs with size > 0 for deletion. + WalGcMode::Regular => wal_sst.metadata.size > 0, + // In fence mode, only consider zero-byte WAL SSTs for deletion. + WalGcMode::Fence => wal_sst.metadata.size == 0, + }) + .filter(|wal_sst| { + Self::is_wal_sst_eligible_for_deletion( + &utc_now, + wal_sst, + &min_age, + &referenced_ranges, + ) + }) + .collect::>(); + let ssts_to_delete = retain_allowed_by_gc_filter(&self.gc_filter, ssts_to_delete).await; + let sst_ids_to_delete = ssts_to_delete + .into_iter() + .map(|wal_sst| wal_sst.id) + .collect::>(); + + self.maybe_delete_wal_ssts(sst_ids_to_delete).await; + + Ok(()) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::block_cache_policy::BlockCachePolicy; + use crate::format::sst::SsTableFormat; + use crate::object_stores::ObjectStores; + use crate::tablestore::TableStoreKind; + use crate::RowEntry; + use object_store::memory::InMemory; + use object_store::path::Path; + use object_store::ObjectStore; + use slatedb_common::clock::MockSystemClock; + use slatedb_common::metrics::MetricsRecorderHelper; + use std::time::Duration; + + fn build_table_store() -> Arc { + let object_store: Arc = Arc::new(InMemory::new()); + Arc::new(TableStore::new( + ObjectStores::new(object_store, None), + SsTableFormat::default(), + Path::from("/"), + None, + TableStoreKind::GC, + BlockCachePolicy::default(), + )) + } + + fn build_collector( + table_store: Arc, + clock: Arc, + mode: WalGcMode, + min_age: Duration, + ) -> SlateDbWalGc { + SlateDbWalGc::new( + table_store, + Arc::new(GcStats::new(&MetricsRecorderHelper::noop())), + GarbageCollectorDirectoryOptions { + interval: None, + min_age, + dry_run: false, + }, + mode, + None, + clock, + ) + } + + async fn write_regular_wal(table_store: &Arc, wal_id: u64) { + let mut sst = table_store.wal_table_builder(); + sst.add(RowEntry::new_value(b"key", b"value", wal_id)) + .await + .unwrap(); + let sst = sst.build().await.unwrap(); + table_store + .write_sst(&SsTableId::Wal(wal_id), &sst) + .await + .unwrap(); + } + + async fn write_fence_wal(table_store: &Arc, wal_id: u64) { + table_store.write_wal_fence(wal_id).await.unwrap(); + } + + async fn wal_ids(table_store: &Arc) -> Vec { + table_store + .list_wal_ssts(..) + .await + .unwrap() + .into_iter() + .map(|wal| wal.id.unwrap_wal_id()) + .collect() + } + + async fn make_all_wals_older_than( + table_store: &Arc, + clock: &MockSystemClock, + min_age: Duration, + ) { + let newest_wal = table_store + .list_wal_ssts(..) + .await + .unwrap() + .into_iter() + .map(|wal| wal.metadata.last_modified) + .max() + .expect("expected at least one WAL"); + let min_age_millis = + i64::try_from(min_age.as_millis()).expect("min_age should fit in i64 milliseconds"); + clock.set(newest_wal.timestamp_millis() + min_age_millis + 1_000); + } + + fn protect_outer_wals() -> Vec { + vec![ + WalFileRange(Bound::Included(1), Bound::Excluded(2)), + WalFileRange(Bound::Included(4), Bound::Unbounded), + ] + } + + #[tokio::test] + async fn regular_mode_deletes_unreferenced_range_and_keeps_referenced_wals() { + let table_store = build_table_store(); + let clock = Arc::new(MockSystemClock::new()); + for wal_id in 1..=4 { + write_regular_wal(&table_store, wal_id).await; + } + make_all_wals_older_than(&table_store, &clock, Duration::ZERO).await; + let collector = build_collector( + table_store.clone(), + clock, + WalGcMode::Regular, + Duration::ZERO, + ); + + collector.collect(protect_outer_wals()).await.unwrap(); + + assert_eq!(wal_ids(&table_store).await, vec![1, 4]); + } + + #[tokio::test] + async fn regular_mode_does_not_touch_fence_wals() { + let table_store = build_table_store(); + let clock = Arc::new(MockSystemClock::new()); + write_regular_wal(&table_store, 1).await; + write_fence_wal(&table_store, 2).await; + make_all_wals_older_than(&table_store, &clock, Duration::ZERO).await; + let collector = build_collector( + table_store.clone(), + clock, + WalGcMode::Regular, + Duration::ZERO, + ); + + collector.collect(vec![]).await.unwrap(); + + assert_eq!(wal_ids(&table_store).await, vec![2]); + } + + #[tokio::test] + async fn regular_mode_respects_min_age() { + let table_store = build_table_store(); + let clock = Arc::new(MockSystemClock::new()); + write_regular_wal(&table_store, 1).await; + let last_modified = table_store + .metadata(&SsTableId::Wal(1)) + .await + .unwrap() + .last_modified; + let min_age = Duration::from_secs(60 * 60); + let collector = build_collector( + table_store.clone(), + clock.clone(), + WalGcMode::Regular, + min_age, + ); + + clock.set((last_modified + chrono::Duration::minutes(30)).timestamp_millis()); + collector.collect(vec![]).await.unwrap(); + assert_eq!(wal_ids(&table_store).await, vec![1]); + + clock.set((last_modified + chrono::Duration::minutes(61)).timestamp_millis()); + collector.collect(vec![]).await.unwrap(); + assert!(wal_ids(&table_store).await.is_empty()); + } + + #[tokio::test] + async fn fence_mode_deletes_unreferenced_range_and_keeps_referenced_wals() { + let table_store = build_table_store(); + let clock = Arc::new(MockSystemClock::new()); + for wal_id in 1..=4 { + write_fence_wal(&table_store, wal_id).await; + } + make_all_wals_older_than(&table_store, &clock, Duration::ZERO).await; + let collector = + build_collector(table_store.clone(), clock, WalGcMode::Fence, Duration::ZERO); + + collector.collect(protect_outer_wals()).await.unwrap(); + + assert_eq!(wal_ids(&table_store).await, vec![1, 4]); + } + + #[tokio::test] + async fn fence_mode_does_not_touch_regular_wals() { + let table_store = build_table_store(); + let clock = Arc::new(MockSystemClock::new()); + write_fence_wal(&table_store, 1).await; + write_regular_wal(&table_store, 2).await; + make_all_wals_older_than(&table_store, &clock, Duration::ZERO).await; + let collector = + build_collector(table_store.clone(), clock, WalGcMode::Fence, Duration::ZERO); + + collector.collect(vec![]).await.unwrap(); + + assert_eq!(wal_ids(&table_store).await, vec![2]); + } + + #[tokio::test] + async fn fence_mode_respects_min_age() { + let table_store = build_table_store(); + let clock = Arc::new(MockSystemClock::new()); + write_fence_wal(&table_store, 1).await; + let last_modified = table_store + .metadata(&SsTableId::Wal(1)) + .await + .unwrap() + .last_modified; + let min_age = Duration::from_secs(60 * 60); + let collector = build_collector( + table_store.clone(), + clock.clone(), + WalGcMode::Fence, + min_age, + ); + + clock.set((last_modified + chrono::Duration::minutes(30)).timestamp_millis()); + collector.collect(vec![]).await.unwrap(); + assert_eq!(wal_ids(&table_store).await, vec![1]); + + clock.set((last_modified + chrono::Duration::minutes(61)).timestamp_millis()); + collector.collect(vec![]).await.unwrap(); + assert!(wal_ids(&table_store).await.is_empty()); + } +} diff --git a/slatedb/src/wal/mod.rs b/slatedb/src/wal/mod.rs index 14b5d6eb7b..aebf4d4d04 100644 --- a/slatedb/src/wal/mod.rs +++ b/slatedb/src/wal/mod.rs @@ -8,6 +8,7 @@ use std::fmt::{Display, Formatter}; use std::ops::{Bound, Range}; use std::sync::Arc; +pub(crate) mod gc; #[cfg(test)] pub(crate) mod test_utils; pub(crate) mod wal_disabled; @@ -15,6 +16,7 @@ pub(crate) mod wal_sst_builder; pub(crate) mod writer_init; /// A range of WAL File IDs +#[derive(Clone, Debug, Eq, PartialEq)] pub struct WalFileRange(pub Bound, pub Bound); impl From> for WalFileRange { @@ -266,6 +268,18 @@ pub trait WalReader { ) -> Result, WalError>; } +/// Trait that defines the contract between SlateDB's garbage collector and a custom WAL +/// implementation. SlateDB tracks the set of currently referenced WAL ranges in its manifest. +/// When the Garbage Collector runs, it computes this set and calls [`WalGC::collect`] so that +/// the implementation can clean up any un-referenced WAL storage. +#[async_trait] +pub trait WalGC: Send + Sync + 'static { + /// Hook for garbage collecting the WAL. Takes a list of ranges of WAL Files that are currently + /// referenced by some active Manifest. The implementation may delete any WAL File that is not + /// included in the ranges in this list. + async fn collect(&self, referenced_ranges: Vec) -> Result<(), WalError>; +} + impl From for WalError { fn from(status: WalStatus) -> Self { status From d0c3d635b6d5d69053f52fe15bdd3ee4487e28ed Mon Sep 17 00:00:00 2001 From: Chris Date: Tue, 4 Aug 2026 10:55:28 -0700 Subject: [PATCH 17/65] fix(compactor): prevent terminal compaction resurrection (#2002) --- slatedb/src/compactor.rs | 57 +++++++++++----- slatedb/src/compactor_state.rs | 31 +++++---- slatedb/src/compactor_state_protocols.rs | 87 +++++++++++++++++++++++- 3 files changed, 143 insertions(+), 32 deletions(-) diff --git a/slatedb/src/compactor.rs b/slatedb/src/compactor.rs index e26f29c42f..7806b61bab 100644 --- a/slatedb/src/compactor.rs +++ b/slatedb/src/compactor.rs @@ -635,8 +635,18 @@ impl CompactorEventHandler { .active_compactions() .filter(|c| c.status() != CompactionStatus::Compacted) { - let estimated_source_bytes = - Self::calculate_estimated_source_bytes(compaction, db_state); + let Some(estimated_source_bytes) = + Self::calculate_estimated_source_bytes(compaction, db_state) + else { + warn!( + "skipping compaction progress because a source is absent from the manifest \ + [id={}, status={:?}, spec={}]", + compaction.id(), + compaction.status(), + compaction.spec(), + ); + continue; + }; total_estimated_bytes += estimated_source_bytes; total_bytes_processed += compaction.bytes_processed(); @@ -690,10 +700,11 @@ impl CompactorEventHandler { } /// Calculates the estimated total source bytes for a compaction. - fn calculate_estimated_source_bytes(compaction: &Compaction, db_state: &ManifestCore) -> u64 { - let tree = db_state - .tree_for_segment(compaction.spec().segment()) - .expect("compaction target segment missing from manifest"); + fn calculate_estimated_source_bytes( + compaction: &Compaction, + db_state: &ManifestCore, + ) -> Option { + let tree = db_state.tree_for_segment(compaction.spec().segment())?; let views_by_id: HashMap = tree.l0.iter().map(|view| (view.id, view)).collect(); @@ -704,17 +715,13 @@ impl CompactorEventHandler { .spec() .sources() .iter() - .map(|source| match source { - SourceId::SstView(id) => views_by_id - .get(id) - .expect("compaction source view not found in L0") - .estimate_size(), - SourceId::SortedRun(id) => srs_by_id - .get(id) - .expect("compaction source sorted run not found") - .estimate_size(), + .try_fold(0, |total, source| { + let source_bytes = match source { + SourceId::SstView(id) => views_by_id.get(id)?.estimate_size(), + SourceId::SortedRun(id) => srs_by_id.get(id)?.estimate_size(), + }; + Some(total + source_bytes) }) - .sum() } /// Handles a polling tick by refreshing compactions and the manifest, then possibly scheduling compactions. @@ -3870,7 +3877,23 @@ mod tests { let expected = segment_l0.estimate_size() + segment_sr.estimate_size(); let actual = CompactorEventHandler::calculate_estimated_source_bytes(&compaction, &core); - assert_eq!(actual, expected); + assert_eq!(actual, Some(expected)); + } + + #[test] + fn test_calculate_estimated_source_bytes_returns_none_for_missing_source() { + let missing_l0 = Ulid::new(); + let compaction = Compaction::new( + Ulid::new(), + CompactionSpec::new(vec![SourceId::SstView(missing_l0)], 1), + ); + + let actual = CompactorEventHandler::calculate_estimated_source_bytes( + &compaction, + &ManifestCore::new(), + ); + + assert_eq!(actual, None); } #[tokio::test] diff --git a/slatedb/src/compactor_state.rs b/slatedb/src/compactor_state.rs index b9e42bb28d..4e5382cfdd 100644 --- a/slatedb/src/compactor_state.rs +++ b/slatedb/src/compactor_state.rs @@ -678,9 +678,9 @@ impl CompactionsCore { self } - /// Returns an iterator over all recent compactions. Recent compactions include all - /// active (submitted or running) compactions as well as the most recently finished - /// compaction (failed or completed). + /// Returns all compactions retained in this state. Persisted state contains all active + /// compactions and the most recently finished one. Process-local state may also retain + /// older terminal entries until its next successful write. pub(crate) fn recent_compactions(&self) -> impl Iterator { self.recent_compactions.values() } @@ -821,7 +821,8 @@ impl Compactions { /// /// This is the in-memory view that a single compactor task uses to: /// - keep a fresh `DirtyManifest` (view of `CoreDbState`), -/// - track in-flight compactions by id (ULID). +/// - track in-flight compactions by id (ULID), and +/// - retain terminal tombstones until their `.compactions` write succeeds. pub struct CompactorState { manifest: DirtyObject, compactions: DirtyObject, @@ -918,11 +919,11 @@ impl CompactorState { // For compactions not in local state (Vacant), accept new submissions, // worker-completed results, and retained terminal entries. On a write // conflict, the retry path reloads and merges the persisted `.compactions` - // object before writing again. At that point, local state may already have - // committed or pruned an entry while persisted state still contains an older - // Compacted/Completed/Failed view. Insert those entries, let the commit path - // resolve stale Compacted entries via `validate_compaction`, and let the compactions - // write path resolve stale terminal entries via `retain_active_and_last_finished`. + // object before writing again. Process-local terminal entries are deliberately + // retained until that write succeeds, so an older remote state for the same id + // is treated as a stale transition rather than resurrected as active work. + // Insert genuinely absent entries and let the commit path resolve stale + // Compacted entries via `validate_compaction`. // // Scheduled/Running are different: they carry no finished output and require prior // coordinator ownership, so seeing them absent from local state remains anomalous. @@ -960,11 +961,10 @@ impl CompactorState { } } - let mut merged_compactions = Compactions { + let merged_compactions = Compactions { compactor_epoch: self.compactions.value.compactor_epoch, core: CompactionsCore::new().with_compactions(merged), }; - merged_compactions.retain_active_and_last_finished(); remote_compactions.value = merged_compactions; self.set_compactions(remote_compactions); } @@ -1050,7 +1050,13 @@ impl CompactorState { Ok(()) } - /// Mutates a compaction in place if it exists, then trims retained state. + /// Mutates a compaction in place if it exists. + /// + /// Terminal entries remain in process-local state until the next successful + /// `.compactions` write. They act as tombstones during conflict retries so + /// an older persisted `Submitted` or `Compacted` record cannot be resurrected. + /// The persisted value is still trimmed by + /// [`crate::compactor_state_protocols::CompactorStateWriter`]. pub(crate) fn update_compaction(&mut self, compaction_id: &Ulid, f: F) where F: FnOnce(&mut Compaction), @@ -1064,7 +1070,6 @@ impl CompactorState { { f(compaction); } - self.compactions.value.retain_active_and_last_finished(); } /// Applies the effects of a finished compaction to the in-memory manifest. diff --git a/slatedb/src/compactor_state_protocols.rs b/slatedb/src/compactor_state_protocols.rs index 142b58fce9..8005158127 100644 --- a/slatedb/src/compactor_state_protocols.rs +++ b/slatedb/src/compactor_state_protocols.rs @@ -290,6 +290,11 @@ impl CompactorStateWriter { /// Persists the current compactions state to the compactions store and refreshes the /// local dirty object with the latest version. /// + /// Process-local state retains every terminal transition until this write succeeds. + /// Only the outgoing value is trimmed to the latest terminal entry. This preserves + /// terminal entries as tombstones across conflict retries while keeping the persisted + /// `.compactions` object bounded. + /// /// ## Returns /// - `Ok(())` when compactions are successfully written. /// - `SlateDBError` if an unrecoverable error occurs. @@ -307,8 +312,10 @@ impl CompactorStateWriter { } Err(err) if err.is_sequenced_write_conflict() => { // Merge the latest remote state (e.g. a worker's Compacted write) into - // the coordinator's view before retrying. Without this, retrying with a stale - // desired_value could silently overwrite worker progress. + // the coordinator's untrimmed local view before retrying. Without this, + // retrying with a stale desired_value could silently overwrite worker + // progress. Local terminal entries also prevent stale remote active states + // for the same ids from being resurrected. self.load_compactions().await?; desired_value = self.state.compactions().value.clone(); desired_value.retain_active_and_last_finished(); @@ -890,6 +897,82 @@ mod tests { assert_eq!(final_id, start_id + 2); } + #[tokio::test] + async fn test_write_compactions_safely_does_not_resurrect_terminal_on_conflict() { + let object_store: Arc = Arc::new(InMemory::new()); + let manifest_store = Arc::new(ManifestStore::new( + &Path::from(ROOT), + Arc::clone(&object_store), + )); + let compactions_store = Arc::new(CompactionsStore::new( + &Path::from(ROOT), + Arc::clone(&object_store), + )); + let system_clock: Arc = Arc::new(DefaultSystemClock::new()); + + StoredManifest::create_new_db( + manifest_store.clone(), + ManifestCore::new(), + system_clock.clone(), + ) + .await + .unwrap(); + + let mut writer = CompactorStateWriter::new( + manifest_store, + compactions_store.clone(), + system_clock, + &CompactorOptions::default(), + Arc::new(DbRand::new(7)), + ) + .await + .unwrap(); + + let completed_id = Ulid::from_parts(10, 0); + let failed_id = Ulid::from_parts(20, 0); + let spec = CompactionSpec::new(vec![], 0); + writer + .state + .insert_compaction_for_test(Compaction::new(completed_id, spec.clone())); + writer + .state + .insert_compaction_for_test(Compaction::new(failed_id, spec.clone())); + writer.state.update_compaction(&completed_id, |compaction| { + compaction.set_status(CompactionStatus::Completed) + }); + writer.state.update_compaction(&failed_id, |compaction| { + compaction.set_status(CompactionStatus::Failed) + }); + + // Advance the remote version with the pre-transition state of the completed + // compaction. This forces the coordinator's first write to conflict and reload. + let mut external = StoredCompactions::load(compactions_store.clone()) + .await + .unwrap(); + let mut dirty = external.prepare_dirty().unwrap(); + dirty.value.insert(Compaction::new(completed_id, spec)); + external.update(dirty).await.unwrap(); + + writer.write_compactions_safely().await.unwrap(); + + let persisted = compactions_store.read_latest_compactions().await.unwrap(); + assert!( + persisted + .recent_compactions() + .all(|compaction| compaction.id() != completed_id), + "stale Submitted compaction was resurrected" + ); + assert_eq!( + persisted + .recent_compactions() + .find(|compaction| compaction.id() == failed_id) + .expect("latest terminal compaction was not retained") + .status(), + CompactionStatus::Failed + ); + assert_eq!(writer.state.active_compactions().count(), 0); + } + #[tokio::test] async fn test_write_compactions_safely_retries_on_boundary_conflict() { let object_store: Arc = Arc::new(InMemory::new()); From aa46b9b8296fc4d0d0586b137818f6b666f9c37b Mon Sep 17 00:00:00 2001 From: Aditya Mishra Date: Tue, 4 Aug 2026 23:33:09 +0530 Subject: [PATCH 18/65] Add cleanup_db admin fn and cli command for clone teardown (#2001) --- slatedb-cli/src/args.rs | 9 + slatedb-cli/src/main.rs | 15 ++ slatedb/src/admin.rs | 496 +++++++++++++++++++++++++++++++++++++++- 3 files changed, 519 insertions(+), 1 deletion(-) diff --git a/slatedb-cli/src/args.rs b/slatedb-cli/src/args.rs index 838a7fd35b..3603121ce8 100644 --- a/slatedb-cli/src/args.rs +++ b/slatedb-cli/src/args.rs @@ -128,6 +128,15 @@ pub(crate) enum CliCommands { id: Uuid, }, + /// Delete a database: strip any checkpoints it pinned in parent databases, + /// then delete its own objects. Without --confirm, prints what it would + /// delete and does nothing. + DeleteDb { + /// Actually delete. Without it, this is a dry run. + #[arg(long)] + confirm: bool, + }, + /// List the current checkpoints of the db. ListCheckpoints { /// Optionally specify the name to filter the checkpoints. Note that name may not be unique diff --git a/slatedb-cli/src/main.rs b/slatedb-cli/src/main.rs index 08d5fd44ef..fd63d3cd9b 100644 --- a/slatedb-cli/src/main.rs +++ b/slatedb-cli/src/main.rs @@ -66,6 +66,7 @@ async fn main() -> Result<(), Box> { exec_refresh_checkpoint(&admin, id, lifetime).await?; } CliCommands::DeleteCheckpoint { id } => exec_delete_checkpoint(&admin, id).await?, + CliCommands::DeleteDb { confirm } => exec_delete_db(&admin, confirm).await?, CliCommands::ListCheckpoints { name } => exec_list_checkpoints(&admin, name).await?, CliCommands::RunGarbageCollection { resource, @@ -280,6 +281,20 @@ async fn exec_delete_checkpoint(admin: &Admin, id: Uuid) -> Result<(), Box Result<(), Box> { + let paths = admin.delete_db(confirm).await?; + if confirm { + println!("deleted {} objects", paths.len()); + } else { + println!("would delete {} objects:", paths.len()); + for path in &paths { + println!(" {path}"); + } + println!("rerun with --confirm to delete"); + } + Ok(()) +} + async fn exec_list_checkpoints( admin: &Admin, name_filter: Option, diff --git a/slatedb/src/admin.rs b/slatedb/src/admin.rs index 8ba01bc1d7..38e322f6f6 100644 --- a/slatedb/src/admin.rs +++ b/slatedb/src/admin.rs @@ -19,8 +19,9 @@ use crate::seq_tracker::FindOption; use crate::utils::IdGenerator; use bytes::Bytes; use chrono::{DateTime, Utc}; +use futures::StreamExt; use object_store::path::Path; -use object_store::ObjectStore; +use object_store::{ObjectStore, ObjectStoreExt}; use rand::RngCore; use slatedb_common::DbRand; use std::env; @@ -559,6 +560,117 @@ impl Admin { .map_err(Into::into) } + /// Deletes a database, stripping any checkpoints it pinned in parent + /// databases (a clone) before removing its own objects. Works for plain and + /// cloned dbs alike: a plain db just has no parent checkpoints to strip. + /// + /// Without `confirm` this is a dry run: it returns every object it *would* + /// delete and touches nothing. Pass `confirm` to actually delete. + /// + /// The delete writes a `.deleting` marker while the manifest still proves + /// this is a slatedb dir, then removes everything else, then the marker. If a + /// prior run crashed mid-delete, the leftover marker lets a rerun finish the + /// job. A `confirm` delete of a dir with neither a manifest nor a marker is + /// refused, so a fat-fingered `--path` can't wipe an unrelated directory. + /// Idempotent. + pub async fn delete_db(&self, confirm: bool) -> Result, crate::Error> { + let main = self.retrying_store(ObjectStoreType::Main); + + if !confirm { + return self.list_prefix(&main).await; + } + + let marker = self.path.clone().join(".deleting"); + let marker_exists = main.get(&marker).await.map(|_| true).or_else(|e| match e { + object_store::Error::NotFound { .. } => Ok(false), + other => Err(SlateDBError::from(other)), + })?; + + let manifest = self.manifest_store().try_read_latest_manifest().await?; + + // No manifest and no marker means we never proved this is a slatedb dir. + // If there are objects under the prefix, it may be a fat-fingered path we + // must not wipe, so refuse. If empty, there's nothing to delete anyway + // (also the already-deleted no-op), so fall through to a clean return. + if manifest.is_none() + && !marker_exists + && !collect_prefix(&main, &self.path).await?.is_empty() + { + return Err(SlateDBError::InvalidDBState.into()); + } + + // Strip the checkpoints this db pinned in each parent. Needs the + // manifest, which a resumed (marker-only) run may no longer have. + if let Some(manifest) = manifest.as_ref() { + for external_db in manifest.external_dbs() { + let Some(final_checkpoint_id) = external_db.final_checkpoint_id else { + continue; + }; + let parent_store = Arc::new(ManifestStore::new( + &Path::from(external_db.path.as_str()), + self.retrying_store(ObjectStoreType::Main), + )); + let mut parent = + match StoredManifest::load(parent_store, self.system_clock.clone()).await { + Ok(parent) => parent, + // parent already deleted: no checkpoint left to strip, skip it + Err(SlateDBError::LatestTransactionalObjectVersionMissing) => continue, + Err(e) => return Err(e.into()), + }; + parent.delete_checkpoint(final_checkpoint_id).await?; + } + } + + // Commit the intent to delete while the manifest still proves this dir. + if !marker_exists { + main.put(&marker, Bytes::new().into()) + .await + .map_err(SlateDBError::from)?; + } + + // Delete everything but the marker, then the marker last, so a crash in + // between leaves the marker to prove a rerun should finish. + let mut deleted = self.delete_prefix(&main, Some(&marker)).await?; + if self.object_stores.has_wal_object_store() { + deleted.extend( + self.delete_prefix(&self.retrying_store(ObjectStoreType::Wal), None) + .await?, + ); + } + main.delete(&marker).await.map_err(SlateDBError::from)?; + deleted.push(marker); + Ok(deleted) + } + + /// Lists every object under this db's path prefix across the main and WAL stores. + async fn list_prefix(&self, main: &Arc) -> Result, crate::Error> { + let mut paths = collect_prefix(main, &self.path).await?; + if self.object_stores.has_wal_object_store() { + paths.extend( + collect_prefix(&self.retrying_store(ObjectStoreType::Wal), &self.path).await?, + ); + } + Ok(paths) + } + + /// Deletes every object under this db's path prefix in the given store, + /// skipping `keep` if set. Returns the deleted paths. + async fn delete_prefix( + &self, + store: &Arc, + keep: Option<&Path>, + ) -> Result, crate::Error> { + let mut deleted = Vec::new(); + for path in collect_prefix(store, &self.path).await? { + if Some(&path) == keep { + continue; + } + store.delete(&path).await.map_err(SlateDBError::from)?; + deleted.push(path); + } + Ok(deleted) + } + /// Returns the timestamp or sequence from the latest manifest's sequence tracker. /// When `round_up` is true, uses the next higher value; otherwise the previous one. pub async fn get_timestamp_for_sequence( @@ -815,6 +927,24 @@ pub fn load_gcp() -> Result, crate::Error> { })?) as Arc) } +/// Collects every object path under `prefix` in the given store. +async fn collect_prefix( + store: &Arc, + prefix: &Path, +) -> Result, crate::Error> { + let mut listing = store.list(Some(prefix)); + let mut paths = Vec::new(); + while let Some(meta) = listing + .next() + .await + .transpose() + .map_err(SlateDBError::from)? + { + paths.push(meta.location); + } + Ok(paths) +} + #[cfg(test)] mod tests { use crate::admin::{load_object_store_from_env, AdminBuilder}; @@ -1389,6 +1519,291 @@ mod tests { assert!(manifest.is_ok(), "cloned manifest should exist"); } + #[tokio::test] + async fn test_delete_db_removes_checkpoint_from_parent() { + use crate::admin::CloneSourceSpec; + use crate::config::CheckpointOptions; + use crate::manifest::store::{ManifestStore, StoredManifest}; + use crate::Db; + + let object_store: Arc = Arc::new(InMemory::new()); + let system_clock = Arc::new(DefaultSystemClock::new()); + let parent_path = Path::from("/tmp/test_cleanup_parent"); + let clone_path = Path::from("/tmp/test_cleanup_clone"); + + let parent_db = Db::open(parent_path.clone(), object_store.clone()) + .await + .unwrap(); + parent_db.close().await.unwrap(); + + // An unrelated checkpoint in the parent that cleanup must not touch. + let parent_admin = AdminBuilder::new(parent_path.clone(), object_store.clone()).build(); + let unrelated = parent_admin + .create_detached_checkpoint(&CheckpointOptions::default()) + .await + .unwrap() + .id; + + let clone_admin = AdminBuilder::new(clone_path.clone(), object_store.clone()).build(); + clone_admin + .create_clone_builder_from_source(CloneSourceSpec::new(parent_path.clone())) + .build() + .await + .expect("clone should succeed"); + + // The checkpoint the clone pinned in the parent. + let clone_ms = Arc::new(ManifestStore::new(&clone_path, object_store.clone())); + let clone_stored = StoredManifest::load(clone_ms, system_clock.clone()) + .await + .unwrap(); + let pinned = clone_stored.manifest().external_dbs[0] + .final_checkpoint_id + .expect("clone pins a final_checkpoint_id in the parent"); + + let read_parent_checkpoints = || { + let object_store = object_store.clone(); + let system_clock = system_clock.clone(); + let parent_path = parent_path.clone(); + async move { + let ms = Arc::new(ManifestStore::new(&parent_path, object_store)); + let stored = StoredManifest::load(ms, system_clock).await.unwrap(); + stored + .manifest() + .core + .checkpoints + .iter() + .map(|c| c.id) + .collect::>() + } + }; + + let before = read_parent_checkpoints().await; + assert!( + before.contains(&pinned), + "parent should have pinned checkpoint before cleanup" + ); + assert!( + before.contains(&unrelated), + "parent should have unrelated checkpoint" + ); + + clone_admin + .delete_db(true) + .await + .expect("delete should succeed"); + + let after = read_parent_checkpoints().await; + assert!( + !after.contains(&pinned), + "pinned checkpoint should be gone from parent" + ); + assert!( + after.contains(&unrelated), + "unrelated checkpoint should remain" + ); + } + + #[tokio::test] + async fn test_delete_db_deletes_clone_after_parent_already_gone() { + use crate::admin::CloneSourceSpec; + use crate::Db; + use futures::StreamExt; + use object_store::ObjectStoreExt; + + let object_store: Arc = Arc::new(InMemory::new()); + let parent_path = Path::from("/tmp/test_delete_orphan_parent"); + let clone_path = Path::from("/tmp/test_delete_orphan_clone"); + + Db::open(parent_path.clone(), object_store.clone()) + .await + .unwrap() + .close() + .await + .unwrap(); + + let clone_admin = AdminBuilder::new(clone_path.clone(), object_store.clone()).build(); + clone_admin + .create_clone_builder_from_source(CloneSourceSpec::new(parent_path.clone())) + .build() + .await + .expect("clone should succeed"); + + // Wipe the parent out from under the clone. The clone still names it in + // external_dbs with a pinned checkpoint, but the parent manifest is gone. + let mut parent_listing = object_store.list(Some(&parent_path)); + while let Some(meta) = parent_listing.next().await { + object_store.delete(&meta.unwrap().location).await.unwrap(); + } + + // A missing parent means there is no checkpoint left to strip, so the + // clone must still be deletable rather than wedged on a Data error. + clone_admin + .delete_db(true) + .await + .expect("clone should delete even with parent already gone"); + assert_eq!( + object_store.list(Some(&clone_path)).count().await, + 0, + "clone objects should be gone" + ); + } + + #[tokio::test] + async fn test_delete_db_deletes_own_objects_only_with_confirm() { + use crate::admin::CloneSourceSpec; + use crate::Db; + use futures::StreamExt; + + let object_store: Arc = Arc::new(InMemory::new()); + let parent_path = Path::from("/tmp/test_delete_confirm_parent"); + let clone_path = Path::from("/tmp/test_delete_confirm_clone"); + + Db::open(parent_path.clone(), object_store.clone()) + .await + .unwrap() + .close() + .await + .unwrap(); + + let clone_admin = AdminBuilder::new(clone_path.clone(), object_store.clone()).build(); + clone_admin + .create_clone_builder_from_source(CloneSourceSpec::new(parent_path.clone())) + .build() + .await + .expect("clone should succeed"); + + let count_under = |prefix: Path| { + let object_store = object_store.clone(); + async move { object_store.list(Some(&prefix)).count().await } + }; + + let initial = count_under(clone_path.clone()).await; + assert!(initial > 0, "clone should have objects"); + + // confirm = false is a dry run: it reports what it would delete and + // touches nothing. + let would_delete = clone_admin + .delete_db(false) + .await + .expect("dry run should succeed"); + assert_eq!( + would_delete.len(), + initial, + "dry run should report every object under the prefix" + ); + assert_eq!( + count_under(clone_path.clone()).await, + initial, + "dry run must not delete anything" + ); + + // confirm = true deletes the clone's own objects. + clone_admin + .delete_db(true) + .await + .expect("delete should succeed"); + assert_eq!( + count_under(clone_path.clone()).await, + 0, + "clone objects should be gone after confirm" + ); + + // Idempotent: a second run over an already-deleted db is a clean no-op. + clone_admin + .delete_db(true) + .await + .expect("second delete should be a no-op"); + } + + #[tokio::test] + async fn test_delete_db_finishes_partial_deletion_via_marker() { + use crate::admin::CloneSourceSpec; + use crate::Db; + use futures::StreamExt; + use object_store::ObjectStoreExt; + + let object_store: Arc = Arc::new(InMemory::new()); + let parent_path = Path::from("/tmp/test_delete_partial_parent"); + let clone_path = Path::from("/tmp/test_delete_partial_clone"); + + Db::open(parent_path.clone(), object_store.clone()) + .await + .unwrap() + .close() + .await + .unwrap(); + + let clone_admin = AdminBuilder::new(clone_path.clone(), object_store.clone()).build(); + clone_admin + .create_clone_builder_from_source(CloneSourceSpec::new(parent_path.clone())) + .build() + .await + .expect("clone should succeed"); + + // Simulate a crash mid-delete: the marker was written, then the manifests + // (and some objects) were removed, but the marker and other objects remain. + object_store + .put( + &clone_path.clone().join(".deleting"), + bytes::Bytes::new().into(), + ) + .await + .unwrap(); + let manifest_prefix = Path::from("/tmp/test_delete_partial_clone/manifest"); + let mut listing = object_store.list(Some(&manifest_prefix)); + while let Some(meta) = listing.next().await { + object_store.delete(&meta.unwrap().location).await.unwrap(); + } + let count_under = |prefix: Path| { + let object_store = object_store.clone(); + async move { object_store.list(Some(&prefix)).count().await } + }; + assert!( + count_under(clone_path.clone()).await > 0, + "leftover clone objects should remain after partial deletion" + ); + + // The marker proves this is a real slatedb dir, so delete resumes and + // finishes the job even though the manifest is already gone. + clone_admin + .delete_db(true) + .await + .expect("delete should finish partial deletion"); + assert_eq!( + count_under(clone_path.clone()).await, + 0, + "leftover objects, including the marker, should be gone" + ); + } + + #[tokio::test] + async fn test_delete_db_refuses_without_manifest_or_marker() { + use futures::StreamExt; + use object_store::ObjectStoreExt; + + let object_store: Arc = Arc::new(InMemory::new()); + let dir = Path::from("/tmp/test_delete_fat_finger"); + + // A directory that is NOT a slatedb dir: no manifest, no .deleting marker. + // This stands in for a fat-fingered --path. + object_store + .put( + &dir.clone().join("important.txt"), + bytes::Bytes::from_static(b"keepme").into(), + ) + .await + .unwrap(); + + let admin = AdminBuilder::new(dir.clone(), object_store.clone()).build(); + admin + .delete_db(true) + .await + .expect_err("delete must refuse a dir with no manifest and no marker"); + + let count = object_store.list(Some(&dir)).count().await; + assert_eq!(count, 1, "the unrelated object must be left untouched"); + } + #[cfg(feature = "wal_disable")] #[tokio::test] async fn test_create_clone_with_multiple_sources() { @@ -1471,4 +1886,83 @@ mod tests { "clone should have an external database for each parent" ); } + + #[cfg(feature = "wal_disable")] + #[tokio::test] + async fn test_delete_db_removes_checkpoints_from_all_parents() { + use crate::config::{PutOptions, Settings, WriteOptions}; + use crate::manifest::store::{ManifestStore, StoredManifest}; + use crate::{admin::CloneSourceSpec, Db}; + use uuid::Uuid; + + let object_store: Arc = Arc::new(InMemory::new()); + let system_clock = Arc::new(DefaultSystemClock::new()); + let parent_path1 = Path::from("/tmp/test_cleanup_multi_parent1"); + let parent_path2 = Path::from("/tmp/test_cleanup_multi_parent2"); + let clone_path = Path::from("/tmp/test_cleanup_multi_clone"); + + let settings = Settings { + wal_enabled: false, + ..Settings::default() + }; + let write_opts = WriteOptions::default(); + + // Two parents with disjoint single-key SSTs (the union path rejects overlaps). + for (path, key) in [(&parent_path1, b"a"), (&parent_path2, b"z")] { + let db = Db::builder(path.clone(), object_store.clone()) + .with_settings(settings.clone()) + .build() + .await + .unwrap(); + db.put_with_options(key, b"1", &PutOptions::default(), &write_opts) + .await + .unwrap(); + db.close().await.unwrap(); + } + + let clone_admin = AdminBuilder::new(clone_path.clone(), object_store.clone()).build(); + clone_admin + .create_clone_builder_from_source(CloneSourceSpec::new(parent_path1.clone())) + .with_source(CloneSourceSpec::new(parent_path2.clone())) + .build() + .await + .expect("clone with multiple sources should succeed"); + + // Collect the checkpoint each parent got pinned with. + let clone_ms = Arc::new(ManifestStore::new(&clone_path, object_store.clone())); + let clone_stored = StoredManifest::load(clone_ms, system_clock.clone()) + .await + .unwrap(); + let pinned: Vec<(String, Uuid)> = clone_stored + .manifest() + .external_dbs + .iter() + .map(|e| (e.path.clone(), e.final_checkpoint_id.unwrap())) + .collect(); + assert_eq!(pinned.len(), 2); + + clone_admin + .delete_db(true) + .await + .expect("delete should succeed"); + + for (parent_path, checkpoint_id) in pinned { + let ms = Arc::new(ManifestStore::new( + &parent_path.into(), + object_store.clone(), + )); + let stored = StoredManifest::load(ms, system_clock.clone()) + .await + .unwrap(); + assert!( + !stored + .manifest() + .core + .checkpoints + .iter() + .any(|c| c.id == checkpoint_id), + "pinned checkpoint should be removed from every parent" + ); + } + } } From 0717cc1e4e9bad10a4773760f66bac4264ecf05e Mon Sep 17 00:00:00 2001 From: Chris Date: Tue, 4 Aug 2026 14:05:19 -0700 Subject: [PATCH 19/65] Run nightly pprof on a Warp ARM runner (#2008) --- .github/workflows/nightly.yaml | 13 +++++++++---- 1 file changed, 9 insertions(+), 4 deletions(-) diff --git a/.github/workflows/nightly.yaml b/.github/workflows/nightly.yaml index b3b509d808..62b9ee9091 100644 --- a/.github/workflows/nightly.yaml +++ b/.github/workflows/nightly.yaml @@ -206,7 +206,7 @@ jobs: echo "Total diagrams: $(ls -1 target/bencher/transaction-results/mermaid/*.mermaid 2>/dev/null | wc -l)" microbenchmark-pprofs: - runs-on: ubuntu-latest + runs-on: warp-ubuntu-latest-arm64-8x steps: - uses: actions/checkout@v4 @@ -234,10 +234,15 @@ jobs: # install instructions from https://github.com/polarsignals/pprofme/tree/main - name: Install pprofme run: | - curl -LO https://github.com/polarsignals/pprofme/releases/latest/download/pprofme_$(uname)_$(uname -m) + PPROFME_ARCH=$(uname -m) + if [ "$PPROFME_ARCH" = "aarch64" ]; then + PPROFME_ARCH=arm64 + fi + PPROFME_BINARY="pprofme_$(uname)_${PPROFME_ARCH}" + curl -fLO "https://github.com/polarsignals/pprofme/releases/latest/download/${PPROFME_BINARY}" curl -sL https://github.com/polarsignals/pprofme/releases/latest/download/pprofme_checksums.txt | shasum --ignore-missing -a 256 --check - chmod a+x pprofme_$(uname)_$(uname -m) - sudo mv pprofme_$(uname)_$(uname -m) /usr/local/bin/pprofme + chmod a+x "${PPROFME_BINARY}" + sudo mv "${PPROFME_BINARY}" /usr/local/bin/pprofme - name: Upload pprofs and add links to summary run: | From 889d405eabba655e76905dcbe2fab2c748eb9092 Mon Sep 17 00:00:00 2001 From: Rui Fan <1996fanrui@gmail.com> Date: Wed, 5 Aug 2026 18:22:50 +0200 Subject: [PATCH 20/65] [1972] Expose manifest version count as a metric (#1974) --- .../src/garbage_collector/compactions_gc.rs | 189 ++++++++++++++++-- slatedb/src/garbage_collector/manifest_gc.rs | 185 ++++++++++++++++- slatedb/src/garbage_collector/stats.rs | 13 +- 3 files changed, 364 insertions(+), 23 deletions(-) diff --git a/slatedb/src/garbage_collector/compactions_gc.rs b/slatedb/src/garbage_collector/compactions_gc.rs index f4634dd2c0..326e448bd5 100644 --- a/slatedb/src/garbage_collector/compactions_gc.rs +++ b/slatedb/src/garbage_collector/compactions_gc.rs @@ -25,6 +25,7 @@ use crate::{ use chrono::{DateTime, Utc}; use futures::StreamExt; use log::error; +use std::sync::atomic::{AtomicU64, Ordering}; use std::sync::Arc; use super::filter::retain_allowed_by_gc_filter; @@ -72,7 +73,9 @@ impl CompactionsGcTask { /// Deletes the given compactions files from the compactions store. /// /// In case of dryrun, the actual deletion doesn't happen. - async fn maybe_delete_compactions(&self, compactions_ids: Vec) { + /// + /// Returns the number of compactions files actually deleted. + async fn maybe_delete_compactions(&self, compactions_ids: Vec) -> u64 { if self.compactions_options.dry_run { if !compactions_ids.is_empty() { log::info!( @@ -86,22 +89,28 @@ impl CompactionsGcTask { id ); } - return; + return 0; } + let deleted_count = AtomicU64::new(0); futures::stream::iter(compactions_ids) - .for_each_concurrent(GC_DELETE_CONCURRENCY, |id| async move { - if let Err(e) = self - .compactions_store - .delete_compactions_unchecked(id) - .await - { - error!("error deleting compactions [id={:?}, error={}]", id, e); - } else { - self.stats.gc_compactions_count.increment(1); + .for_each_concurrent(GC_DELETE_CONCURRENCY, |id| { + let deleted_count = &deleted_count; + async move { + if let Err(e) = self + .compactions_store + .delete_compactions_unchecked(id) + .await + { + error!("error deleting compactions [id={:?}, error={}]", id, e); + } else { + self.stats.gc_compactions_count.increment(1); + deleted_count.fetch_add(1, Ordering::Relaxed); + } } }) .await; + deleted_count.load(Ordering::Relaxed) } } @@ -112,6 +121,7 @@ impl GcTask for CompactionsGcTask { async fn collect(&self, utc_now: DateTime) -> Result<(), SlateDBError> { let min_age = self.compactions_min_age(); let mut compactions_metadata_list = self.compactions_store.list_compactions(..).await?; + let pre_gc_count = compactions_metadata_list.len() as u64; // Remove the last element so we never delete the latest compactions file compactions_metadata_list.pop(); @@ -142,9 +152,18 @@ impl GcTask for CompactionsGcTask { .map(|compactions_metadata| compactions_metadata.id) .collect::>(); - self.maybe_delete_compactions(compactions_ids_to_delete) + self.stats.gc_compactions_versions.set(pre_gc_count as i64); + + let deleted_count = self + .maybe_delete_compactions(compactions_ids_to_delete) .await; + if deleted_count > 0 { + self.stats + .gc_compactions_versions + .set((pre_gc_count - deleted_count) as i64); + } + Ok(()) } @@ -160,7 +179,9 @@ mod tests { use async_trait::async_trait; use chrono::TimeDelta; use object_store::{memory::InMemory, path::Path, ObjectStoreExt}; - use slatedb_common::metrics::MetricsRecorderHelper; + use slatedb_common::metrics::{ + lookup_metric_with_labels, DefaultMetricsRecorder, MetricsRecorderHelper, + }; use slatedb_common::ObjectMetadata; use std::collections::HashSet; use std::time::Duration; @@ -337,4 +358,146 @@ mod tests { .unwrap() .is_some()); } + + async fn make_compactions_store() -> (Arc, StoredCompactions) { + let object_store = Arc::new(InMemory::new()); + let compactions_store = Arc::new(CompactionsStore::new(&Path::from("/root"), object_store)); + let stored = StoredCompactions::create(compactions_store.clone(), 0) + .await + .unwrap(); + (compactions_store, stored) + } + + #[tokio::test] + async fn test_version_count_after_gc_deletes_old_compactions() { + let (compactions_store, mut stored) = make_compactions_store().await; + // Write two more compactions files: ids 1 (create), 2, 3 + stored + .update(stored.prepare_dirty().unwrap()) + .await + .unwrap(); + stored + .update(stored.prepare_dirty().unwrap()) + .await + .unwrap(); + + let metrics = Arc::new(DefaultMetricsRecorder::new()); + let recorder = MetricsRecorderHelper::new(metrics.clone(), Default::default()); + let task = CompactionsGcTask::new( + compactions_store.clone(), + Arc::new(GcStats::new(&recorder)), + GarbageCollectorDirectoryOptions { + min_age: Duration::ZERO, + interval: None, + dry_run: false, + }, + None, + true, + ); + + task.collect(Utc::now() + TimeDelta::hours(1)) + .await + .unwrap(); + + // GC deletes compactions 1 and 2 (older than min_age=0); compactions 3 survives as latest. + assert_eq!( + lookup_metric_with_labels( + &metrics, + crate::garbage_collector::stats::VERSION_COUNT, + &[("resource", "compactions")] + ), + Some(1), + "expected 1 surviving compactions file after GC" + ); + } + + #[tokio::test] + async fn test_version_count_when_nothing_to_delete() { + let (compactions_store, mut stored) = make_compactions_store().await; + stored + .update(stored.prepare_dirty().unwrap()) + .await + .unwrap(); + stored + .update(stored.prepare_dirty().unwrap()) + .await + .unwrap(); + + let metrics = Arc::new(DefaultMetricsRecorder::new()); + let recorder = MetricsRecorderHelper::new(metrics.clone(), Default::default()); + let task = CompactionsGcTask::new( + compactions_store.clone(), + Arc::new(GcStats::new(&recorder)), + GarbageCollectorDirectoryOptions { + min_age: Duration::from_secs(3600), // too new to delete + interval: None, + dry_run: false, + }, + None, + true, + ); + + task.collect(Utc::now()).await.unwrap(); + + // Nothing deleted — all 3 compactions files survive. + assert_eq!( + lookup_metric_with_labels( + &metrics, + crate::garbage_collector::stats::VERSION_COUNT, + &[("resource", "compactions")] + ), + Some(3), + "expected all 3 compactions files when nothing qualifies for deletion" + ); + } + + #[tokio::test] + async fn test_version_count_unchanged_on_dry_run() { + let (compactions_store, mut stored) = make_compactions_store().await; + // Write two more compactions files: ids 1 (create), 2, 3 + stored + .update(stored.prepare_dirty().unwrap()) + .await + .unwrap(); + stored + .update(stored.prepare_dirty().unwrap()) + .await + .unwrap(); + + let metrics = Arc::new(DefaultMetricsRecorder::new()); + let recorder = MetricsRecorderHelper::new(metrics.clone(), Default::default()); + let task = CompactionsGcTask::new( + compactions_store.clone(), + Arc::new(GcStats::new(&recorder)), + GarbageCollectorDirectoryOptions { + min_age: Duration::ZERO, + interval: None, + dry_run: true, + }, + None, + true, + ); + + task.collect(Utc::now() + TimeDelta::hours(1)) + .await + .unwrap(); + + // Dry run deletes nothing, so all 3 compactions files still exist and the gauge + // reports the true current count rather than the hypothetical post-deletion count. + assert_eq!( + lookup_metric_with_labels( + &metrics, + crate::garbage_collector::stats::VERSION_COUNT, + &[("resource", "compactions")] + ), + Some(3), + "expected dry run to leave the gauge at the true current count" + ); + let compactions = compactions_store.list_compactions(..).await.unwrap(); + assert_eq!( + compactions.len(), + 3, + "dry run should not delete any compactions files" + ); + } } diff --git a/slatedb/src/garbage_collector/manifest_gc.rs b/slatedb/src/garbage_collector/manifest_gc.rs index 709ec3c80c..5e1cfa9d8f 100644 --- a/slatedb/src/garbage_collector/manifest_gc.rs +++ b/slatedb/src/garbage_collector/manifest_gc.rs @@ -5,6 +5,7 @@ use chrono::{DateTime, Utc}; use futures::StreamExt; use log::error; use std::collections::HashSet; +use std::sync::atomic::{AtomicU64, Ordering}; use std::sync::Arc; use super::filter::retain_allowed_by_gc_filter; @@ -52,7 +53,9 @@ impl ManifestGcTask { /// Deletes the given manifests from the manifest store. /// /// In case of dryrun, the actual deletion doesn't happen. - async fn maybe_delete_manifests(&self, manifest_ids: Vec) { + /// + /// Returns the number of manifests actually deleted. + async fn maybe_delete_manifests(&self, manifest_ids: Vec) -> u64 { if self.manifest_options.dry_run { if !manifest_ids.is_empty() { log::info!( @@ -63,18 +66,24 @@ impl ManifestGcTask { for id in manifest_ids { log::debug!("dry run: would delete manifest but skipped [id={:?}]", id); } - return; + return 0; } + let deleted_count = AtomicU64::new(0); futures::stream::iter(manifest_ids) - .for_each_concurrent(GC_DELETE_CONCURRENCY, |id| async move { - if let Err(e) = self.manifest_store.delete_manifest_unchecked(id).await { - error!("error deleting manifest [id={:?}, error={}]", id, e); - } else { - self.stats.gc_manifest_count.increment(1); + .for_each_concurrent(GC_DELETE_CONCURRENCY, |id| { + let deleted_count = &deleted_count; + async move { + if let Err(e) = self.manifest_store.delete_manifest_unchecked(id).await { + error!("error deleting manifest [id={:?}, error={}]", id, e); + } else { + self.stats.gc_manifest_count.increment(1); + deleted_count.fetch_add(1, Ordering::Relaxed); + } } }) .await; + deleted_count.load(Ordering::Relaxed) } } @@ -103,6 +112,8 @@ impl GcTask for ManifestGcTask { .collect(); // Delete manifests older than min_age + // Capture length before into_iter() consumes the list; +1 re-adds the popped latest. + let pre_gc_count = manifest_metadata_list.len() as u64 + 1; let manifests_to_delete = manifest_metadata_list .into_iter() .filter(|manifest_metadata| { @@ -131,7 +142,15 @@ impl GcTask for ManifestGcTask { .map(|manifest_metadata| manifest_metadata.id) .collect::>(); - self.maybe_delete_manifests(manifest_ids_to_delete).await; + self.stats.gc_manifest_versions.set(pre_gc_count as i64); + + let deleted_count = self.maybe_delete_manifests(manifest_ids_to_delete).await; + + if deleted_count > 0 { + self.stats + .gc_manifest_versions + .set((pre_gc_count - deleted_count) as i64); + } Ok(()) } @@ -152,7 +171,9 @@ mod tests { use chrono::TimeDelta; use object_store::{memory::InMemory, path::Path, ObjectStoreExt}; use slatedb_common::clock::DefaultSystemClock; - use slatedb_common::metrics::MetricsRecorderHelper; + use slatedb_common::metrics::{ + lookup_metric_with_labels, DefaultMetricsRecorder, MetricsRecorderHelper, + }; use slatedb_common::ObjectMetadata; use std::time::Duration; @@ -329,4 +350,150 @@ mod tests { assert!(manifest_store.try_read_manifest(1).await.unwrap().is_some()); assert!(manifest_store.try_read_manifest(2).await.unwrap().is_some()); } + + async fn make_manifest_store() -> (Arc, StoredManifest) { + let object_store = Arc::new(InMemory::new()); + let manifest_store = Arc::new(ManifestStore::new(&Path::from("/root"), object_store)); + let stored = StoredManifest::create_new_db( + manifest_store.clone(), + ManifestCore::new(), + Arc::new(DefaultSystemClock::new()), + ) + .await + .unwrap(); + (manifest_store, stored) + } + + #[tokio::test] + async fn test_version_count_after_gc_deletes_old_manifests() { + let (manifest_store, mut stored) = make_manifest_store().await; + // Write two more manifests: ids 1 (create), 2, 3 + stored + .update(stored.prepare_dirty().unwrap()) + .await + .unwrap(); + stored + .update(stored.prepare_dirty().unwrap()) + .await + .unwrap(); + + let metrics = Arc::new(DefaultMetricsRecorder::new()); + let recorder = MetricsRecorderHelper::new(metrics.clone(), Default::default()); + let task = ManifestGcTask::new( + manifest_store.clone(), + Arc::new(GcStats::new(&recorder)), + GarbageCollectorDirectoryOptions { + min_age: Duration::ZERO, + interval: None, + dry_run: false, + }, + None, + true, + ); + + task.collect(Utc::now() + TimeDelta::hours(1)) + .await + .unwrap(); + + // GC deletes manifests 1 and 2 (older than min_age=0); manifest 3 survives as latest. + assert_eq!( + lookup_metric_with_labels( + &metrics, + crate::garbage_collector::stats::VERSION_COUNT, + &[("resource", "manifest")] + ), + Some(1), + "expected 1 surviving manifest after GC" + ); + } + + #[tokio::test] + async fn test_version_count_when_nothing_to_delete() { + let (manifest_store, mut stored) = make_manifest_store().await; + stored + .update(stored.prepare_dirty().unwrap()) + .await + .unwrap(); + stored + .update(stored.prepare_dirty().unwrap()) + .await + .unwrap(); + + let metrics = Arc::new(DefaultMetricsRecorder::new()); + let recorder = MetricsRecorderHelper::new(metrics.clone(), Default::default()); + let task = ManifestGcTask::new( + manifest_store.clone(), + Arc::new(GcStats::new(&recorder)), + GarbageCollectorDirectoryOptions { + min_age: Duration::from_secs(3600), // too new to delete + interval: None, + dry_run: false, + }, + None, + true, + ); + + task.collect(Utc::now()).await.unwrap(); + + // Nothing deleted — all 3 manifests survive. + assert_eq!( + lookup_metric_with_labels( + &metrics, + crate::garbage_collector::stats::VERSION_COUNT, + &[("resource", "manifest")] + ), + Some(3), + "expected all 3 manifests when nothing qualifies for deletion" + ); + } + + #[tokio::test] + async fn test_version_count_unchanged_on_dry_run() { + let (manifest_store, mut stored) = make_manifest_store().await; + // Write two more manifests: ids 1 (create), 2, 3 + stored + .update(stored.prepare_dirty().unwrap()) + .await + .unwrap(); + stored + .update(stored.prepare_dirty().unwrap()) + .await + .unwrap(); + + let metrics = Arc::new(DefaultMetricsRecorder::new()); + let recorder = MetricsRecorderHelper::new(metrics.clone(), Default::default()); + let task = ManifestGcTask::new( + manifest_store.clone(), + Arc::new(GcStats::new(&recorder)), + GarbageCollectorDirectoryOptions { + min_age: Duration::ZERO, + interval: None, + dry_run: true, + }, + None, + true, + ); + + task.collect(Utc::now() + TimeDelta::hours(1)) + .await + .unwrap(); + + // Dry run deletes nothing, so all 3 manifests still exist and the gauge reports + // the true current count rather than the hypothetical post-deletion count. + assert_eq!( + lookup_metric_with_labels( + &metrics, + crate::garbage_collector::stats::VERSION_COUNT, + &[("resource", "manifest")] + ), + Some(3), + "expected dry run to leave the gauge at the true current count" + ); + let manifests = manifest_store.list_manifests(..).await.unwrap(); + assert_eq!( + manifests.len(), + 3, + "dry run should not delete any manifests" + ); + } } diff --git a/slatedb/src/garbage_collector/stats.rs b/slatedb/src/garbage_collector/stats.rs index 66fe9ec8f4..c1b25f5702 100644 --- a/slatedb/src/garbage_collector/stats.rs +++ b/slatedb/src/garbage_collector/stats.rs @@ -1,4 +1,4 @@ -use slatedb_common::metrics::{CounterFn, MetricsRecorderHelper}; +use slatedb_common::metrics::{CounterFn, GaugeFn, MetricsRecorderHelper}; use std::sync::Arc; macro_rules! gc_stat_name { @@ -9,6 +9,7 @@ macro_rules! gc_stat_name { pub const DELETED_COUNT: &str = gc_stat_name!("deleted_count"); pub const GC_COUNT: &str = gc_stat_name!("count"); +pub const VERSION_COUNT: &str = gc_stat_name!("version_count"); /// Stats for the garbage collector. pub struct GcStats { @@ -19,6 +20,8 @@ pub struct GcStats { pub gc_compactions_count: Arc, pub gc_detach_count: Arc, pub gc_count: Arc, + pub gc_manifest_versions: Arc, + pub gc_compactions_versions: Arc, } impl GcStats { @@ -49,6 +52,14 @@ impl GcStats { .labels(&[("resource", "detach")]) .register(), gc_count: recorder.counter(GC_COUNT).register(), + gc_manifest_versions: recorder + .gauge(VERSION_COUNT) + .labels(&[("resource", "manifest")]) + .register(), + gc_compactions_versions: recorder + .gauge(VERSION_COUNT) + .labels(&[("resource", "compactions")]) + .register(), } } } From 0f946cbcc854ef70df15536387f59704cc0d98d2 Mon Sep 17 00:00:00 2001 From: Rohan Date: Wed, 5 Aug 2026 14:20:03 -0400 Subject: [PATCH 21/65] [rfc-30 4/N]: move max wal file flush check to wal trait (#2007) --- rfcs/0030-pluggable-wal.md | 16 +++++++++ slatedb/src/batch_write.rs | 66 +++++++++++++++++++--------------- slatedb/src/wal/mod.rs | 12 +++++++ slatedb/src/wal/writer_init.rs | 5 +++ slatedb/src/wal_buffer.rs | 27 ++++++++++++++ 5 files changed, 97 insertions(+), 29 deletions(-) diff --git a/rfcs/0030-pluggable-wal.md b/rfcs/0030-pluggable-wal.md index 94e71c07b3..4d5029db4c 100644 --- a/rfcs/0030-pluggable-wal.md +++ b/rfcs/0030-pluggable-wal.md @@ -331,6 +331,18 @@ pub trait WalWriter: Send { /// future that receives the result of the flush once it completes. async fn flush(&mut self) -> Result; + /// Returns true if the WAL implementation wants to request that the current in-memory + /// writes be flushed to a new l0. WAL implementations can use this to (1) bound the range + /// of writes that need to be replayed when SlateDB restarts, and (2) push data to L0s earlier + /// so that it's available to readers, which poll the latest manifest. + /// + /// ## Arguments + /// - `replay_after_wal_id`: The WAL ID used as the replay point for the last memtable + /// that was flushed to L0 + fn should_flush_memtable(&self, _replay_after_wal_id: u64) -> bool { + false + } + /// Returns a `WalObserver` for reading [`WalStatus`] and subscribing to events. fn observer(&self) -> Box; @@ -557,6 +569,10 @@ Memtable and Db flushing stay the same. The Batch Writer task annotates each imm with a safe replay point using `WalStatus::last_flushed_wal_id`, and flushes the WAL using `WalWriter::flush`. +As part of this change we will move tracking of early memtable flush via +`max_wal_flushes_before_l0_flush` to the native WAL implementation in the implementation of +`WalWriter::should_flush_memtable`. + #### Garbage Collection `WalGcTask` lists manifests to determine the set of referenced WAL File IDs and then delegates diff --git a/slatedb/src/batch_write.rs b/slatedb/src/batch_write.rs index 651ed6d3b3..df264c88d0 100644 --- a/slatedb/src/batch_write.rs +++ b/slatedb/src/batch_write.rs @@ -32,8 +32,6 @@ use futures::{FutureExt, StreamExt}; use std::sync::Arc; use tracing::instrument; -use std::collections::BTreeSet; - use crate::config::WriteOptions; use crate::db_state::DbState; use crate::db_transaction::DbTransaction; @@ -45,6 +43,7 @@ use crate::wal::{FlushResultFuture, WalWriter}; use crate::{batch::WriteBatch, db::DbInner, db::WriteHandle, error::SlateDBError}; use bytes::Bytes; use parking_lot::RwLockWriteGuard; +use std::collections::BTreeSet; use tokio::sync::oneshot; pub(crate) const WRITE_BATCH_TASK_NAME: &str = "writer"; @@ -121,9 +120,10 @@ impl MessageHandler for WriteBatchEventHandler { done, txn, }) => { + let wal_writer = self.wal_writer.as_deref_mut(); let result = self .db_inner - .write_batch(batch, &options, txn.as_ref(), self.wal_writer.as_mut()) + .write_batch(batch, &options, txn.as_ref(), wal_writer) .await; match result { Ok(write_result) => { @@ -189,12 +189,12 @@ impl MessageHandler for WriteBatchEventHandler { impl DbInner { #[allow(clippy::panic)] #[instrument(level = "trace", skip_all, fields(batch_size = batch.op_count()))] - async fn write_batch( - &self, + async fn write_batch<'a>( + &'a self, batch: WriteBatch, options: &WriteOptions, txn: Option<&DbTransaction>, - wal_writer: Option<&mut Box>, + mut wal_writer: Option<&mut (dyn WalWriter + 'static)>, ) -> Result { let _options = options; #[cfg(not(dst))] @@ -252,7 +252,7 @@ impl DbInner { return Ok(Err(error)); } - if let Some(wal_writer) = wal_writer { + if let Some(wal_writer) = wal_writer.as_mut() { assert!(self.wal_enabled); // WAL entries must be appended to the wal buffer atomically. Otherwise, // the WAL buffer might flush the entries in the middle of the batch, which @@ -303,7 +303,7 @@ impl DbInner { self.record_memtable_sequence(commit_seq); // maybe freeze the memtable. - self.maybe_freeze_current_memtable()?; + self.maybe_freeze_current_memtable(wal_writer.as_deref())?; let write_handle = WriteHandle::new_with_waiter(commit_seq, now, self.status_manager.durability_waiter()); @@ -311,31 +311,39 @@ impl DbInner { Ok(Ok(write_handle)) } - fn maybe_freeze_current_memtable(&self) -> Result<(), SlateDBError> { - let replay_after_wal_id = self.wal_observer.status()?.last_flushed_wal_id; - let mut guard = self.state.write(); - let meta = guard.memtable().metadata(); - - let last_freeze_wal_id = guard - .state() - .imm_memtable - .front() - .map(|imm| imm.recent_flushed_wal_id()) - .unwrap_or(guard.state().core().replay_after_wal_id); - - let l0_sst_size_est = self - .table_store - .estimate_encoded_size_compacted(meta.entry_num, meta.entries_size_in_bytes); + fn maybe_freeze_current_memtable( + &self, + wal_writer: Option<&dyn WalWriter>, + ) -> Result<(), SlateDBError> { + // extract the information required to make a freeze decision under a read-lock first + // so we don't call into `WalWriter` while holding a lock + let (l0_sst_size_est, last_freeze_wal_id) = { + let guard = self.state.read(); + let meta = guard.memtable().metadata(); + + let last_freeze_wal_id = guard + .state() + .imm_memtable + .front() + .map(|imm| imm.recent_flushed_wal_id()) + .unwrap_or(guard.state().core().replay_after_wal_id); + + let l0_sst_size_est = self + .table_store + .estimate_encoded_size_compacted(meta.entry_num, meta.entries_size_in_bytes); + + (l0_sst_size_est, last_freeze_wal_id) + }; - let wal_id_gap = replay_after_wal_id - .checked_sub(last_freeze_wal_id) - .ok_or_else(|| SlateDBError::InvalidDBState)?; + let wal_should_flush_memtable = wal_writer + .is_some_and(|wal_writer| wal_writer.should_flush_memtable(last_freeze_wal_id)); - if wal_id_gap < self.settings.max_wal_flushes_before_l0_flush - && l0_sst_size_est < self.settings.l0_sst_size_bytes - { + if !wal_should_flush_memtable && l0_sst_size_est < self.settings.l0_sst_size_bytes { return Ok(()); } + + let replay_after_wal_id = self.wal_observer.status()?.last_flushed_wal_id; + let mut guard = self.state.write(); self.freeze_current_memtable_with_state_guard(&mut guard, replay_after_wal_id); Ok(()) } diff --git a/slatedb/src/wal/mod.rs b/slatedb/src/wal/mod.rs index aebf4d4d04..8efa80e570 100644 --- a/slatedb/src/wal/mod.rs +++ b/slatedb/src/wal/mod.rs @@ -218,6 +218,18 @@ pub trait WalWriter: Send { /// future that receives the result of the flush once it completes. async fn flush(&mut self) -> Result; + /// Returns true if the WAL implementation wants to request that the current in-memory + /// writes be flushed to a new l0. WAL implementations can use this to (1) bound the range + /// of writes that need to be replayed when SlateDB restarts, and (2) push data to L0s earlier + /// so that it's available to readers, which poll the latest manifest. + /// + /// ## Arguments + /// - `replay_after_wal_id`: The WAL ID used as the replay point for the last memtable + /// that was flushed to L0 + fn should_flush_memtable(&self, _replay_after_wal_id: u64) -> bool { + false + } + /// Returns a `WalObserver` for reading [`WalStatus`] and subscribing to events. fn observer(&self) -> Box; diff --git a/slatedb/src/wal/writer_init.rs b/slatedb/src/wal/writer_init.rs index dac03c3968..ddf035deea 100644 --- a/slatedb/src/wal/writer_init.rs +++ b/slatedb/src/wal/writer_init.rs @@ -18,6 +18,7 @@ use std::time::Duration; #[derive(Clone, Copy)] pub(crate) struct WalWriterInitOptions { max_wal_bytes_size: usize, + max_wal_flushes_before_l0_flush: u64, max_flush_interval: Option, } @@ -25,6 +26,7 @@ impl From<&Settings> for WalWriterInitOptions { fn from(settings: &Settings) -> Self { Self { max_wal_bytes_size: settings.l0_sst_size_bytes, + max_wal_flushes_before_l0_flush: settings.max_wal_flushes_before_l0_flush, max_flush_interval: settings.flush_interval, } } @@ -35,6 +37,7 @@ pub(crate) struct WalWriterInit { recorder: MetricsRecorderHelper, table_store: Arc, max_wal_bytes_size: usize, + max_wal_flushes_before_l0_flush: u64, max_flush_interval: Option, empty_wal_id: u64, task_executor: Arc, @@ -61,6 +64,7 @@ impl WalWriterInit { recorder, table_store, max_wal_bytes_size: options.max_wal_bytes_size, + max_wal_flushes_before_l0_flush: options.max_wal_flushes_before_l0_flush, max_flush_interval: options.max_flush_interval, empty_wal_id, task_executor, @@ -135,6 +139,7 @@ impl wal::WriterInit for WalWriterInit { empty_wal_id, self.table_store.clone(), self.max_wal_bytes_size, + self.max_wal_flushes_before_l0_flush, self.max_flush_interval, self.task_executor.clone(), ) diff --git a/slatedb/src/wal_buffer.rs b/slatedb/src/wal_buffer.rs index 72339058ca..c0051409c8 100644 --- a/slatedb/src/wal_buffer.rs +++ b/slatedb/src/wal_buffer.rs @@ -33,6 +33,8 @@ pub(crate) const WAL_BUFFER_TASK_NAME: &str = "wal_writer"; /// /// - `max_wal_size`: Flushes when `max_wal_size` bytes is exceeded /// - `max_flush_interval`: Flushes after `max_flush_interval` elapses, if set +/// - `max_wal_flushes_before_l0_flush`: Requests a memtable flush when this many WAL files have +/// been flushed since the latest memtable replay point /// /// For strict durability requirements on synchronous writes, use [`WalBufferManager::flush()`] to explicitly /// trigger a flush operation and await the result. This will flush ALL the in memory WALs (including the @@ -51,6 +53,7 @@ pub(crate) struct WalBufferManager { stats: Arc, table_store: Arc, max_wal_bytes_size: usize, + max_wal_flushes_before_l0_flush: u64, /// The largest flush_epoch for which a size-triggered flush request has been /// sent. Compared against `flush_epoch` in the inner struct to avoid sending /// redundant flush requests for the same WAL. @@ -111,6 +114,7 @@ impl WalBufferManager { last_flushed_wal_id: u64, table_store: Arc, max_wal_bytes_size: usize, + max_wal_flushes_before_l0_flush: u64, max_flush_interval: Option, task_executor: Arc, ) -> Result { @@ -147,6 +151,7 @@ impl WalBufferManager { stats, table_store, max_wal_bytes_size, + max_wal_flushes_before_l0_flush, last_flush_requested_epoch: AtomicU64::new(0), task_executor, }) @@ -210,6 +215,14 @@ impl WalWriter for WalBufferManager { Ok(()) } + fn should_flush_memtable(&self, replay_after_wal_id: u64) -> bool { + let last_flushed_wal_id = self.inner.read().last_flushed_wal_id; + let Some(wal_id_gap) = last_flushed_wal_id.checked_sub(replay_after_wal_id) else { + return false; + }; + wal_id_gap >= self.max_wal_flushes_before_l0_flush + } + fn observer(&self) -> Box { Box::new(WalObserver { inner: self.inner.clone(), @@ -840,6 +853,7 @@ mod tests { 0, // recent_flushed_wal_id table_store.clone(), 1000, // max_wal_bytes_size + 4096, // max_wal_flushes_before_l0_flush Some(flush_interval), // max_flush_interval task_executor.clone(), ) @@ -861,6 +875,19 @@ mod tests { (wal_buffer, table_store, status_manager, recorder) } + #[tokio::test] + async fn test_should_flush_memtable_at_max_wal_gap() { + let (wal_buffer, _, _, _) = setup_wal_buffer().await; + let max_wal_gap = wal_buffer.max_wal_flushes_before_l0_flush; + let last_flushed_wal_id = max_wal_gap + 10; + wal_buffer.inner.write().last_flushed_wal_id = last_flushed_wal_id; + + assert!(!wal_buffer.should_flush_memtable(last_flushed_wal_id - max_wal_gap + 1)); + assert!(wal_buffer.should_flush_memtable(last_flushed_wal_id - max_wal_gap)); + assert!(wal_buffer.should_flush_memtable(last_flushed_wal_id - max_wal_gap - 1)); + assert!(!wal_buffer.should_flush_memtable(last_flushed_wal_id + 1)); + } + #[tokio::test] async fn test_basic_append_and_flush_operations() { let (mut wal_buffer, table_store, _, _) = setup_wal_buffer().await; From 9e82d15e71f0ad8e41035b5eaa7e582b68082920 Mon Sep 17 00:00:00 2001 From: Thomas B <9094255+Ten0@users.noreply.github.com> Date: Wed, 5 Aug 2026 20:25:08 +0200 Subject: [PATCH 22/65] Actually skip the WAL at DbReader open when skip_wal_replay is set (helps #2003) (#2005) Co-authored-by: Chris --- slatedb/src/config.rs | 9 +- slatedb/src/db_reader.rs | 571 ++++++++++++++++++++++++++++----------- 2 files changed, 411 insertions(+), 169 deletions(-) diff --git a/slatedb/src/config.rs b/slatedb/src/config.rs index 19e99a2905..2150f3ee08 100644 --- a/slatedb/src/config.rs +++ b/slatedb/src/config.rs @@ -1111,13 +1111,12 @@ pub struct DbReaderOptions { /// local filesystem, mirroring the behaviour of `Db`. pub object_store_cache_options: ObjectStoreCacheOptions, - /// When true, skip WAL replay entirely. The reader will only see data that has been - /// compacted into L0 or lower levels. This is useful for read-heavy workloads that - /// don't need to see the most recent uncommitted writes and want to minimize the + /// When true, skip WAL replay entirely, in every reader mode. The reader reads no WAL + /// when it opens or when it refreshes its state, so it only sees data that has been + /// flushed to L0 or lower levels. This is useful for read-heavy workloads that + /// don't need to see the most recent writes and want to minimize the /// cost of opening many readers. /// - /// WAL replay is also skipped in [`crate::DbReaderMode::Checkpoint`] mode. - /// /// When combined with a reader mode that polls manifests, the reader will still see newly /// compacted data as manifests are updated. /// diff --git a/slatedb/src/db_reader.rs b/slatedb/src/db_reader.rs index 12e4c21bb1..ee98efb388 100644 --- a/slatedb/src/db_reader.rs +++ b/slatedb/src/db_reader.rs @@ -1,46 +1,50 @@ -use crate::bytes_range::{ByteRangeBounds, BytesRange}; -use crate::cached_object_store::CachedObjectStore; -use crate::clock::MonotonicClock; -use crate::config::{CheckpointOptions, DbReaderOptions, ReadOptions, ScanOptions}; -use crate::db_cache::CacheTarget; -use crate::db_cache_manager; -use crate::db_common::extract_segment_prefix; -use crate::db_state::{collect_touched_segments, SsTableId}; -use crate::db_stats::DbStats; -use crate::db_status::{ClosedResultWriter, DbStatus, DbStatusManager}; -use crate::dispatcher::{MessageHandler, MessageHandlerExecutor, MessageTickerDef}; -use crate::error::SlateDBError; -use crate::iter::IterationOrder; -use crate::manifest::store::{ManifestStore, StoredManifest}; -use crate::manifest::{Manifest, ManifestCore, VersionedManifest}; -use crate::mem_table::{ImmutableMemtable, KVTable, WritableKVTable}; -use crate::merge_operator::MergeOperatorType; -use crate::oracle::DbReaderOracle; -use crate::paths::PathResolver; -use crate::prefix_extractor::PrefixExtractor; -use crate::reader::{DbStateReader, Reader, ScanContext}; -use crate::sst_iter::SstIteratorOptions; -use crate::tablestore::TableStore; -use crate::types::KeyValue; -use crate::utils::IdGenerator; -use crate::wal_replay::{WalIteratorOptions, WalReplayIterator, WalReplayOptions}; -use crate::{Checkpoint, DbIterator}; -use crate::{DbCacheManagerOps, DbMetadataOps, DbReadOps}; -use async_trait::async_trait; -use bytes::Bytes; -use futures::stream::BoxStream; -use log::{info, warn}; -use object_store::path::Path; -use object_store::ObjectStore; -use parking_lot::RwLock; -use slatedb_common::clock::SystemClock; -use slatedb_common::DbRand; -use std::collections::{BTreeSet, VecDeque}; -use std::ops::Sub; -use std::sync::Arc; -use std::sync::LazyLock; -use tokio::runtime::Handle; -use uuid::Uuid; +use { + crate::{ + bytes_range::{ByteRangeBounds, BytesRange}, + cached_object_store::CachedObjectStore, + clock::MonotonicClock, + config::{CheckpointOptions, DbReaderOptions, ReadOptions, ScanOptions}, + db_cache::CacheTarget, + db_cache_manager, + db_common::extract_segment_prefix, + db_state::{collect_touched_segments, SsTableId}, + db_stats::DbStats, + db_status::{ClosedResultWriter, DbStatus, DbStatusManager}, + dispatcher::{MessageHandler, MessageHandlerExecutor, MessageTickerDef}, + error::SlateDBError, + iter::IterationOrder, + manifest::{ + store::{ManifestStore, StoredManifest}, + Manifest, ManifestCore, VersionedManifest, + }, + mem_table::{ImmutableMemtable, KVTable, WritableKVTable}, + merge_operator::MergeOperatorType, + oracle::DbReaderOracle, + paths::PathResolver, + prefix_extractor::PrefixExtractor, + reader::{DbStateReader, Reader, ScanContext}, + sst_iter::SstIteratorOptions, + tablestore::TableStore, + types::KeyValue, + utils::IdGenerator, + wal_replay::{WalIteratorOptions, WalReplayIterator, WalReplayOptions}, + Checkpoint, DbCacheManagerOps, DbIterator, DbMetadataOps, DbReadOps, + }, + async_trait::async_trait, + bytes::Bytes, + futures::stream::BoxStream, + log::{info, warn}, + object_store::{path::Path, ObjectStore}, + parking_lot::RwLock, + slatedb_common::{clock::SystemClock, DbRand}, + std::{ + collections::{BTreeSet, VecDeque}, + ops::Sub, + sync::{Arc, LazyLock}, + }, + tokio::runtime::Handle, + uuid::Uuid, +}; pub(crate) const DB_READER_TASK_NAME: &str = "manifest_poller"; @@ -68,6 +72,39 @@ pub enum DbReaderMode { FollowLatest, } +/// Where a reader stops replaying the WAL when it builds its state. +/// +/// This is only reached when replay is wanted at all; a reader configured with +/// [`DbReaderOptions::skip_wal_replay`] reads no WAL, which +/// [`WalReplayEnd::for_reader`] expresses as `None`. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +enum WalReplayEnd { + /// Stop at the manifest's `next_wal_sst_id`, replaying exactly the WAL that + /// the manifest itself records as durable. + Manifest, + + /// Probe the object store for the newest WAL file and replay through it, + /// picking up writes made after the manifest was written. + Latest, +} + +impl WalReplayEnd { + /// Returns `None` when the reader is configured to skip WAL replay, in which + /// case it observes only the state recorded in the manifest (L0 and below). + fn for_reader(mode: DbReaderMode, options: &DbReaderOptions) -> Option { + if options.skip_wal_replay { + return None; + } + Some(match mode { + // A pinned checkpoint reads the state its manifest captured, so it + // stops at that manifest's WAL boundary instead of following WAL + // files written after the checkpoint was taken. + DbReaderMode::Checkpoint(_) => Self::Manifest, + DbReaderMode::ManagedCheckpoint | DbReaderMode::FollowLatest => Self::Latest, + }) + } +} + /// Read-only interface for accessing a database from either /// the latest persistent state or from an arbitrary checkpoint. pub struct DbReader { @@ -154,15 +191,13 @@ impl DbReaderInner { } else { (manifest.id(), manifest.manifest().clone()) }; - let replay_new_wals = - !matches!(mode, DbReaderMode::Checkpoint(_)) && !options.skip_wal_replay; let initial_state = Arc::new( Self::build_reader_state( checkpoint, manifest_id, initial_manifest, VecDeque::new(), - replay_new_wals, + WalReplayEnd::for_reader(mode, &options), Arc::clone(&table_store), &options, segment_extractor.as_ref(), @@ -175,8 +210,8 @@ impl DbReaderInner { initial_state.core().last_l0_clock_tick, )); - // initial_state contains the last_committed_seq after WAL replay. in no-wal mode, we can simply fallback - // to last_l0_seq. + // initial_state contains the last_committed_seq after WAL replay. in no-wal mode, we can + // simply fallback to last_l0_seq. let initial_durable_seq = initial_state .last_remote_persisted_seq .max(initial_state.core().last_l0_seq); @@ -363,7 +398,7 @@ impl DbReaderInner { &self.options, current_state.core(), &mut imm_memtable, - true, + WalReplayEnd::Latest, self.segment_extractor.as_ref(), ) .await?; @@ -433,7 +468,7 @@ impl DbReaderInner { manifest_id, manifest, imm_memtable, - !self.options.skip_wal_replay, + WalReplayEnd::for_reader(self.mode, &self.options), Arc::clone(&self.table_store), &self.options, self.segment_extractor.as_ref(), @@ -446,20 +481,27 @@ impl DbReaderInner { manifest_id: u64, manifest: Manifest, mut imm_memtable: VecDeque>, - replay_new_wals: bool, + replay_wals: Option, table_store: Arc, options: &DbReaderOptions, segment_extractor: Option<&Arc>, ) -> Result { - let (last_wal_id, last_committed_seq) = Self::replay_wal_into( - Arc::clone(&table_store), - options, - &manifest.core, - &mut imm_memtable, - replay_new_wals, - segment_extractor, - ) - .await?; + let (last_wal_id, last_committed_seq) = match replay_wals { + Some(replay_end) => { + Self::replay_wal_into( + Arc::clone(&table_store), + options, + &manifest.core, + &mut imm_memtable, + replay_end, + segment_extractor, + ) + .await? + } + // Skipping replay reads no WAL at all: the reader stays at the + // watermark it has already reached (the most recently read manifest) + None => Self::replayed_watermark(&manifest.core, &imm_memtable), + }; Ok(ReaderState { manifest_id, @@ -577,12 +619,28 @@ impl DbReaderInner { result } + /// The `(last replayed WAL id, last committed seq)` the reader has already + /// reached: the watermark of the most recently replayed table, or the + /// manifest's own boundary when nothing has been replayed into `tables`. + fn replayed_watermark( + core: &ManifestCore, + tables: &VecDeque>, + ) -> (u64, u64) { + match tables.front() { + Some(latest_replayed_table) => ( + latest_replayed_table.recent_flushed_wal_id(), + latest_replayed_table.table().last_seq().unwrap_or(0), + ), + None => (core.replay_after_wal_id, core.last_l0_seq), + } + } + async fn replay_wal_into( table_store: Arc, reader_options: &DbReaderOptions, core: &ManifestCore, into_tables: &mut VecDeque>, - replay_new_wals: bool, + replay_end: WalReplayEnd, segment_extractor: Option<&Arc>, ) -> Result<(u64, u64), SlateDBError> { let sst_iter_options = SstIteratorOptions { @@ -597,18 +655,10 @@ impl DbReaderInner { }; let (mut replay_after_wal_id, mut last_committed_seq) = - if let Some(latest_replayed_table) = into_tables.front() { - ( - latest_replayed_table.recent_flushed_wal_id(), - latest_replayed_table.table().last_seq().unwrap_or(0), - ) - } else { - (core.replay_after_wal_id, core.last_l0_seq) - }; - let wal_id_end = if replay_new_wals { - table_store.last_seen_wal_id(replay_after_wal_id).await? + 1 - } else { - core.next_wal_sst_id + Self::replayed_watermark(core, into_tables); + let wal_id_end = match replay_end { + WalReplayEnd::Manifest => core.next_wal_sst_id, + WalReplayEnd::Latest => table_store.last_seen_wal_id(replay_after_wal_id).await? + 1, }; let iterator_options = WalIteratorOptions { @@ -617,7 +667,8 @@ impl DbReaderInner { }; let replay_options = WalReplayOptions { max_memtable_bytes: reader_options.max_memtable_bytes as usize, - // Skip entries that we already have in `imm_memtable` (that might be above last_l0_seq). + // Skip entries that we already have in `imm_memtable` (that might be above + // last_l0_seq). min_seq: Some(last_committed_seq), }; @@ -695,8 +746,8 @@ impl DbReaderInner { /// /// ## Returns /// - `Ok(())` if the reader is still open. - /// - `Err(SlateDBError::Closed)` if the reader was closed successfully - /// (state.result_reader() returns Ok(())). + /// - `Err(SlateDBError::Closed)` if the reader was closed successfully (state.result_reader() + /// returns Ok(())). /// - `Err(e)` if the reader was closed with an error, where `e` is the error /// (state.result_reader() returns Err(e)). pub(crate) fn check_closed(&self) -> Result<(), SlateDBError> { @@ -864,9 +915,13 @@ impl DbReader { /// # Examples /// /// ``` - /// use slatedb::{Db, DbReader, Error}; - /// use slatedb::object_store::{ObjectStore, memory::InMemory}; - /// use std::sync::Arc; + /// use { + /// slatedb::{ + /// object_store::{memory::InMemory, ObjectStore}, + /// Db, DbReader, Error, + /// }, + /// std::sync::Arc, + /// }; /// /// #[tokio::main] /// async fn main() -> Result<(), Error> { @@ -875,9 +930,7 @@ impl DbReader { /// let db = Db::open("test_db", Arc::clone(&object_store)).await?; /// db.close().await?; /// // Then open a reader - /// let reader = DbReader::builder("test_db", object_store) - /// .build() - /// .await?; + /// let reader = DbReader::builder("test_db", object_store).build().await?; /// Ok(()) /// } /// ``` @@ -964,9 +1017,14 @@ impl DbReader { /// ## Examples /// /// ``` - /// use slatedb::{Db, DbReader, DbReaderMode, config::DbReaderOptions, Error}; - /// use slatedb::object_store::{ObjectStore, memory::InMemory}; - /// use std::sync::Arc; + /// use { + /// slatedb::{ + /// config::DbReaderOptions, + /// object_store::{memory::InMemory, ObjectStore}, + /// Db, DbReader, DbReaderMode, Error, + /// }, + /// std::sync::Arc, + /// }; /// /// #[tokio::main] /// async fn main() -> Result<(), Error> { @@ -976,11 +1034,12 @@ impl DbReader { /// db.flush().await?; /// /// let reader = DbReader::open( - /// "test_db", - /// Arc::clone(&object_store), - /// DbReaderMode::ManagedCheckpoint, - /// DbReaderOptions::default(), - /// ).await?; + /// "test_db", + /// Arc::clone(&object_store), + /// DbReaderMode::ManagedCheckpoint, + /// DbReaderOptions::default(), + /// ) + /// .await?; /// assert_eq!(reader.get(b"key").await?, Some("value".into())); /// Ok(()) /// } @@ -1012,9 +1071,14 @@ impl DbReader { /// ## Examples /// /// ``` - /// use slatedb::{Db, DbReader, DbReaderMode, config::DbReaderOptions, config::ReadOptions, Error}; - /// use slatedb::object_store::{ObjectStore, memory::InMemory}; - /// use std::sync::Arc; + /// use { + /// slatedb::{ + /// config::{DbReaderOptions, ReadOptions}, + /// object_store::{memory::InMemory, ObjectStore}, + /// Db, DbReader, DbReaderMode, Error, + /// }, + /// std::sync::Arc, + /// }; /// /// #[tokio::main] /// async fn main() -> Result<(), Error> { @@ -1024,12 +1088,16 @@ impl DbReader { /// db.flush().await?; /// /// let reader = DbReader::open( - /// "test_db", - /// Arc::clone(&object_store), - /// DbReaderMode::ManagedCheckpoint, - /// DbReaderOptions::default(), - /// ).await?; - /// assert_eq!(db.get_with_options(b"key", &ReadOptions::default()).await?, Some("value".into())); + /// "test_db", + /// Arc::clone(&object_store), + /// DbReaderMode::ManagedCheckpoint, + /// DbReaderOptions::default(), + /// ) + /// .await?; + /// assert_eq!( + /// db.get_with_options(b"key", &ReadOptions::default()).await?, + /// Some("value".into()) + /// ); /// Ok(()) /// } /// ``` @@ -1083,9 +1151,14 @@ impl DbReader { /// ## Examples /// /// ``` - /// use slatedb::{Db, DbReader, DbReaderMode, config::DbReaderOptions, Error}; - /// use slatedb::object_store::{ObjectStore, memory::InMemory}; - /// use std::sync::Arc; + /// use { + /// slatedb::{ + /// config::DbReaderOptions, + /// object_store::{memory::InMemory, ObjectStore}, + /// Db, DbReader, DbReaderMode, Error, + /// }, + /// std::sync::Arc, + /// }; /// /// #[tokio::main] /// async fn main() -> Result<(), Error> { @@ -1096,11 +1169,12 @@ impl DbReader { /// db.flush().await?; /// /// let reader = DbReader::open( - /// "test_db", - /// Arc::clone(&object_store), - /// DbReaderMode::ManagedCheckpoint, - /// DbReaderOptions::default(), - /// ).await?; + /// "test_db", + /// Arc::clone(&object_store), + /// DbReaderMode::ManagedCheckpoint, + /// DbReaderOptions::default(), + /// ) + /// .await?; /// let mut iter = reader.scan("a".."b").await?; /// let kv = iter.next().await?.unwrap(); /// assert_eq!(kv.key.as_ref(), b"a"); @@ -1188,8 +1262,8 @@ impl DbReader { /// /// ## Arguments /// - `prefix`: the key prefix to scan - /// - `subrange`: the range of key suffixes (relative to `prefix`) to - /// scan; `..` scans all keys with the prefix + /// - `subrange`: the range of key suffixes (relative to `prefix`) to scan; `..` scans all keys + /// with the prefix /// /// ## Returns /// - `Result`: An iterator with the results of the scan @@ -1212,8 +1286,8 @@ impl DbReader { /// /// ## Arguments /// - `prefix`: the key prefix to scan - /// - `subrange`: the range of key suffixes (relative to `prefix`) to - /// scan; `..` scans all keys with the prefix + /// - `subrange`: the range of key suffixes (relative to `prefix`) to scan; `..` scans all keys + /// with the prefix /// - `options`: the scan options to use /// /// ## Returns @@ -1244,9 +1318,14 @@ impl DbReader { /// ## Examples /// /// ``` - /// use slatedb::{Db, DbReader, DbReaderMode, config::DbReaderOptions, Error}; - /// use slatedb::object_store::{ObjectStore, memory::InMemory}; - /// use std::sync::Arc; + /// use { + /// slatedb::{ + /// config::DbReaderOptions, + /// object_store::{memory::InMemory, ObjectStore}, + /// Db, DbReader, DbReaderMode, Error, + /// }, + /// std::sync::Arc, + /// }; /// /// #[tokio::main] /// async fn main() -> Result<(), Error> { @@ -1258,12 +1337,12 @@ impl DbReader { /// object_store.clone(), /// DbReaderMode::ManagedCheckpoint, /// options, - /// ).await?; + /// ) + /// .await?; /// reader.close().await?; /// Ok(()) /// } /// ``` - /// pub async fn close(&self) -> Result<(), crate::Error> { self.task_executor .shutdown_task(DB_READER_TASK_NAME) @@ -1373,46 +1452,54 @@ impl DbCacheManagerOps for DbReader { #[cfg(test)] mod tests { - use super::{DbReaderMessage, ManifestPoller, ReaderState}; - use crate::block_cache_policy::BlockCachePolicy; - use crate::clock::MonotonicClock; - use crate::config::{ - CheckpointOptions, CheckpointScope, FlushOptions, FlushType, MergeOptions, PutOptions, - Settings, WriteOptions, + use { + super::{DbReaderMessage, ManifestPoller, ReaderState, WalReplayEnd}, + crate::{ + block_cache_policy::BlockCachePolicy, + clock::MonotonicClock, + config::{ + CheckpointOptions, CheckpointScope, CloseOptions, FlushOptions, FlushType, + MergeOptions, PutOptions, Settings, WriteOptions, + }, + db_reader::{DbReader, DbReaderInner, DbReaderMode, DbReaderOptions}, + db_state::{SsTableId, SstType}, + db_stats::DbStats, + db_status::DbStatusManager, + dispatcher::MessageHandler, + error::SlateDBError, + format::sst::SsTableFormat, + iter::IterationOrder, + manifest::{ + store::{ManifestStore, StoredManifest}, + Manifest, ManifestCore, VersionedManifest, + }, + mem_table::{ImmutableMemtable, WritableKVTable}, + merge_operator::MergeOperatorType, + object_stores::ObjectStores, + oracle::DbReaderOracle, + paths::PathResolver, + proptest_util::{rng::new_test_rng, sample}, + reader::Reader, + tablestore::{TableStore, TableStoreKind}, + test_utils, + types::RowEntry, + CloseReason, Db, + }, + bytes::Bytes, + fail_parallel::FailPointRegistry, + object_store::{memory::InMemory, path::Path, ObjectStore, ObjectStoreExt}, + rstest::rstest, + slatedb_common::{ + clock::{DefaultSystemClock, SystemClock}, + DbRand, MockSystemClock, + }, + std::{ + collections::{BTreeMap, VecDeque}, + sync::Arc, + time::Duration, + }, + uuid::Uuid, }; - use crate::db_reader::{DbReader, DbReaderInner, DbReaderMode, DbReaderOptions}; - use crate::db_state::SsTableId; - use crate::db_stats::DbStats; - use crate::db_status::DbStatusManager; - use crate::dispatcher::MessageHandler; - use crate::format::sst::SsTableFormat; - use crate::iter::IterationOrder; - use crate::manifest::store::{ManifestStore, StoredManifest}; - use crate::manifest::{Manifest, ManifestCore, VersionedManifest}; - use crate::mem_table::{ImmutableMemtable, WritableKVTable}; - use crate::merge_operator::MergeOperatorType; - use crate::object_stores::ObjectStores; - use crate::oracle::DbReaderOracle; - use crate::paths::PathResolver; - use crate::proptest_util::rng::new_test_rng; - use crate::proptest_util::sample; - use crate::reader::Reader; - use crate::tablestore::{TableStore, TableStoreKind}; - use crate::types::RowEntry; - use crate::{error::SlateDBError, test_utils, CloseReason, Db}; - use bytes::Bytes; - use fail_parallel::FailPointRegistry; - use object_store::memory::InMemory; - use object_store::path::Path; - use object_store::{ObjectStore, ObjectStoreExt}; - use rstest::rstest; - use slatedb_common::clock::{DefaultSystemClock, SystemClock}; - use slatedb_common::DbRand; - use slatedb_common::MockSystemClock; - use std::collections::{BTreeMap, VecDeque}; - use std::sync::Arc; - use std::time::Duration; - use uuid::Uuid; #[tokio::test] async fn should_get_value_from_db() { @@ -2263,7 +2350,7 @@ mod tests { &DbReaderOptions::default(), &core, &mut into_tables, - false, + WalReplayEnd::Manifest, None, ) .await @@ -2308,7 +2395,7 @@ mod tests { &DbReaderOptions::default(), &core, &mut into_tables, - false, + WalReplayEnd::Manifest, None, ) .await @@ -2363,7 +2450,7 @@ mod tests { &reader_options, &core, &mut into_tables, - false, + WalReplayEnd::Manifest, None, ) .await @@ -2399,7 +2486,7 @@ mod tests { &DbReaderOptions::default(), &core, &mut into_tables, - true, + WalReplayEnd::Latest, None, ) .await @@ -2430,7 +2517,7 @@ mod tests { &DbReaderOptions::default(), &core, &mut into_tables, - true, + WalReplayEnd::Latest, None, ) .await @@ -2476,7 +2563,7 @@ mod tests { &DbReaderOptions::default(), &core, &mut into_tables, - true, + WalReplayEnd::Latest, None, ) .await @@ -2710,6 +2797,160 @@ mod tests { ); } + /// A manifest records the WAL files written since the last L0 flush in + /// `next_wal_sst_id`. Opening a reader must not read them when WAL replay + /// is skipped: that range is exactly the expensive one, since it grows + /// with everything written between L0 flushes. + #[tokio::test] + async fn skip_wal_replay_should_not_read_wals_recorded_in_manifest() { + let recording_store = Arc::new(test_utils::RecordingObjectStore::new(Arc::new( + InMemory::new(), + ))); + let object_store: Arc = recording_store.clone(); + let path = Path::from("/tmp/test_kv_store"); + let test_provider = TestProvider::new(path.clone(), Arc::clone(&object_store)); + + let db = test_provider.new_db(Settings::default()).await.unwrap(); + let flushed_key = b"flushed_key"; + let flushed_value = b"flushed_value"; + db.put(flushed_key, flushed_value).await.unwrap(); + db.flush_with_options(FlushOptions { + flush_type: FlushType::MemTable, + }) + .await + .unwrap(); + + // Write data that stays in the WAL, then close without flushing the + // memtable. Closing persists the manifest, so `next_wal_sst_id` covers + // these WAL files while `replay_after_wal_id` stays at the last L0 flush. + // The write must be awaited to durability first: closing without a + // memtable flush does not flush the WAL, so an unawaited write would + // race the flush interval and might never reach a WAL SST. + let wal_only_key = b"wal_only_key"; + db.put(wal_only_key, b"wal_only_value") + .await + .unwrap() + .await_durable() + .await + .unwrap(); + db.close_with_options(CloseOptions::default().with_flush_memtables(false)) + .await + .unwrap(); + + let core = test_provider + .manifest_store() + .read_latest_manifest() + .await + .unwrap() + .manifest + .core; + assert!( + core.replay_after_wal_id + 1 < core.next_wal_sst_id, + "test needs a manifest that records live WAL files \ + [replay_after_wal_id={}, next_wal_sst_id={}]", + core.replay_after_wal_id, + core.next_wal_sst_id + ); + + recording_store.clear(); + let reader = test_provider + .new_db_reader( + DbReaderOptions { + skip_wal_replay: true, + ..DbReaderOptions::default() + }, + None, + None, + ) + .await + .unwrap(); + + let wal_reads = recording_store + .get_sst_types(false) + .into_iter() + .chain(recording_store.get_sst_types(true)) + .filter(|sst_type| *sst_type == Some(SstType::Wal)) + .count(); + assert_eq!(wal_reads, 0, "reader read WAL SSTs despite skip_wal_replay"); + + assert_eq!(reader.get(wal_only_key).await.unwrap(), None); + assert_eq!( + reader.get(flushed_key).await.unwrap(), + Some(Bytes::from_static(flushed_value)) + ); + } + + /// A checkpoint captures the WAL files that were durable when it was taken, + /// so a pinned reader replays them by default. `skip_wal_replay` opts out of + /// that read, at the cost of not seeing the checkpointed WAL writes. + #[rstest] + #[case(true, None)] + #[case(false, Some(Bytes::from_static(b"wal_only_value")))] + #[tokio::test] + async fn skip_wal_replay_should_control_checkpoint_wal_reads( + #[case] skip_wal_replay: bool, + #[case] expected: Option, + ) { + let object_store: Arc = Arc::new(InMemory::new()); + let path = Path::from("/tmp/test_kv_store"); + let test_provider = TestProvider::new(path.clone(), Arc::clone(&object_store)); + + let db = test_provider.new_db(Settings::default()).await.unwrap(); + db.put(b"flushed_key", b"flushed_value").await.unwrap(); + db.flush_with_options(FlushOptions { + flush_type: FlushType::MemTable, + }) + .await + .unwrap(); + + // This write is only durable in the WAL, so the checkpoint references it + // through the manifest's `next_wal_sst_id` rather than through L0. The + // scope must be `Durable`: `All` would flush the memtable to L0 first, + // leaving the checkpoint with no live WAL. + let wal_only_key = b"wal_only_key"; + db.put(wal_only_key, b"wal_only_value").await.unwrap(); + db.flush_with_options(FlushOptions { + flush_type: FlushType::Wal, + }) + .await + .unwrap(); + let checkpoint = db + .create_checkpoint(CheckpointScope::Durable, &CheckpointOptions::default()) + .await + .unwrap(); + db.close_with_options(CloseOptions::default().with_flush_memtables(false)) + .await + .unwrap(); + + let core = test_provider + .manifest_store() + .read_manifest(checkpoint.manifest_id) + .await + .unwrap() + .core; + assert!( + core.replay_after_wal_id + 1 < core.next_wal_sst_id, + "test needs a checkpoint whose manifest records live WAL files \ + [replay_after_wal_id={}, next_wal_sst_id={}]", + core.replay_after_wal_id, + core.next_wal_sst_id + ); + + let reader = test_provider + .new_db_reader( + DbReaderOptions { + skip_wal_replay, + ..DbReaderOptions::default() + }, + Some(checkpoint.id), + None, + ) + .await + .unwrap(); + + assert_eq!(reader.get(wal_only_key).await.unwrap(), expected); + } + struct TestProvider { object_store: Arc, path: Path, @@ -3066,9 +3307,11 @@ mod tests { // RFC-0024: per-segment compactions, drains, and segment-set changes // are invisible to the root-tree diff. Verify the segments comparison // fires on each of those shapes. - use crate::db_state::{SortedRun, SsTableHandle, SsTableId, SsTableInfo, SsTableView}; - use crate::format::sst::SST_FORMAT_VERSION_LATEST; - use crate::manifest::{LsmTreeState, Segment}; + use crate::{ + db_state::{SortedRun, SsTableHandle, SsTableId, SsTableInfo, SsTableView}, + format::sst::SST_FORMAT_VERSION_LATEST, + manifest::{LsmTreeState, Segment}, + }; fn view(seq: u64) -> SsTableView { SsTableView::identity(SsTableHandle::new( From a494d0ae8a1335cb6a6b30037b902a0a453887ff Mon Sep 17 00:00:00 2001 From: Rohan Date: Wed, 5 Aug 2026 15:10:04 -0400 Subject: [PATCH 23/65] make close flush mode more explicit by using explicit flush type (#2011) --- bindings/go/uniffi/slatedb.go | 127 +++++++++++++++++++++++++++++ bindings/go/uniffi/slatedb.h | 11 +++ bindings/go/uniffi/slatedb_test.go | 32 ++++++++ bindings/uniffi/src/config.rs | 56 ++++++++++++- bindings/uniffi/src/db.rs | 12 ++- bindings/uniffi/src/lib.rs | 2 +- slatedb/src/config.rs | 23 +++--- slatedb/src/db.rs | 71 ++++++++++++---- slatedb/src/db_reader.rs | 11 +-- 9 files changed, 306 insertions(+), 39 deletions(-) diff --git a/bindings/go/uniffi/slatedb.go b/bindings/go/uniffi/slatedb.go index 6f1a944631..a16849adcf 100644 --- a/bindings/go/uniffi/slatedb.go +++ b/bindings/go/uniffi/slatedb.go @@ -957,6 +957,15 @@ func uniffiCheckChecksums() { panic("slatedb: uniffi_slatedb_uniffi_checksum_method_db_shutdown: UniFFI API checksum mismatch") } } + { + checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { + return C.uniffi_slatedb_uniffi_checksum_method_db_shutdown_with_options() + }) + if checksum != 54951 { + // If this happens try cleaning and rebuilding your project + panic("slatedb: uniffi_slatedb_uniffi_checksum_method_db_shutdown_with_options: UniFFI API checksum mismatch") + } + } { checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { return C.uniffi_slatedb_uniffi_checksum_method_db_snapshot() @@ -3333,6 +3342,8 @@ type DbInterface interface { ScanWithOptions(varRange KeyRange, options ScanOptions) (*DbIterator, error) // Flushes outstanding work and closes the database. Shutdown() error + // Performs the requested final flush and closes the database. + ShutdownWithOptions(options CloseOptions) error // Creates a read-only snapshot representing a consistent point in time. Snapshot() (*DbSnapshot, error) // Returns the latest database status snapshot, including the segment @@ -4011,6 +4022,38 @@ func (_self *Db) Shutdown() error { return err } +// Performs the requested final flush and closes the database. +func (_self *Db) ShutdownWithOptions(options CloseOptions) error { + _pointer := _self.ffiObject.incrementPointer("*Db") + defer _self.ffiObject.decrementPointer() + _, err := uniffiRustCallAsync[*Error]( + FfiConverterErrorINSTANCE, + // completeFn + func(handle C.uint64_t, status *C.RustCallStatus) struct{} { + C.ffi_slatedb_uniffi_rust_future_complete_void(handle, status) + return struct{}{} + }, + // liftFn + func(_ struct{}) struct{} { return struct{}{} }, + C.uniffi_slatedb_uniffi_fn_method_db_shutdown_with_options( + _pointer, FfiConverterCloseOptionsINSTANCE.Lower(options)), + // pollFn + func(handle C.uint64_t, continuation C.UniffiRustFutureContinuationCallback, data C.uint64_t) { + C.ffi_slatedb_uniffi_rust_future_poll_void(handle, continuation, data) + }, + // freeFn + func(handle C.uint64_t) { + C.ffi_slatedb_uniffi_rust_future_free_void(handle) + }, + ) + + if err == nil { + return nil + } + + return err +} + // Creates a read-only snapshot representing a consistent point in time. func (_self *Db) Snapshot() (*DbSnapshot, error) { _pointer := _self.ffiObject.incrementPointer("*Db") @@ -9277,6 +9320,49 @@ func (_ FfiDestroyerCloneSourceSpec) Destroy(value CloneSourceSpec) { value.Destroy() } +// Options controlling how a database is shut down. +type CloseOptions struct { + // The final flush to perform before shutdown. When `None`, no final flush is + // triggered and writes that are not durable may be lost. + FlushType *FlushType +} + +func (r *CloseOptions) Destroy() { + FfiDestroyerOptionalFlushType{}.Destroy(r.FlushType) +} + +type FfiConverterCloseOptions struct{} + +var FfiConverterCloseOptionsINSTANCE = FfiConverterCloseOptions{} + +func (c FfiConverterCloseOptions) Lift(rb RustBufferI) CloseOptions { + return LiftFromRustBuffer[CloseOptions](c, rb) +} + +func (c FfiConverterCloseOptions) Read(reader io.Reader) CloseOptions { + return CloseOptions{ + FfiConverterOptionalFlushTypeINSTANCE.Read(reader), + } +} + +func (c FfiConverterCloseOptions) Lower(value CloseOptions) C.RustBuffer { + return LowerIntoRustBuffer[CloseOptions](c, value) +} + +func (c FfiConverterCloseOptions) LowerExternal(value CloseOptions) ExternalCRustBuffer { + return RustBufferFromC(LowerIntoRustBuffer[CloseOptions](c, value)) +} + +func (c FfiConverterCloseOptions) Write(writer io.Writer, value CloseOptions) { + FfiConverterOptionalFlushTypeINSTANCE.Write(writer, value.FlushType) +} + +type FfiDestroyerCloseOptions struct{} + +func (_ FfiDestroyerCloseOptions) Destroy(value CloseOptions) { + value.Destroy() +} + // Canonical compaction record. type Compaction struct { // Compaction ULID string. @@ -13738,6 +13824,47 @@ func (_ FfiDestroyerOptionalFilterContext) Destroy(value *FilterContext) { } } +type FfiConverterOptionalFlushType struct{} + +var FfiConverterOptionalFlushTypeINSTANCE = FfiConverterOptionalFlushType{} + +func (c FfiConverterOptionalFlushType) Lift(rb RustBufferI) *FlushType { + return LiftFromRustBuffer[*FlushType](c, rb) +} + +func (_ FfiConverterOptionalFlushType) Read(reader io.Reader) *FlushType { + if readInt8(reader) == 0 { + return nil + } + temp := FfiConverterFlushTypeINSTANCE.Read(reader) + return &temp +} + +func (c FfiConverterOptionalFlushType) Lower(value *FlushType) C.RustBuffer { + return LowerIntoRustBuffer[*FlushType](c, value) +} + +func (c FfiConverterOptionalFlushType) LowerExternal(value *FlushType) ExternalCRustBuffer { + return RustBufferFromC(LowerIntoRustBuffer[*FlushType](c, value)) +} + +func (_ FfiConverterOptionalFlushType) Write(writer io.Writer, value *FlushType) { + if value == nil { + writeInt8(writer, 0) + } else { + writeInt8(writer, 1) + FfiConverterFlushTypeINSTANCE.Write(writer, *value) + } +} + +type FfiDestroyerOptionalFlushType struct{} + +func (_ FfiDestroyerOptionalFlushType) Destroy(value *FlushType) { + if value != nil { + FfiDestroyerFlushType{}.Destroy(*value) + } +} + type FfiConverterOptionalIterationOrder struct{} var FfiConverterOptionalIterationOrderINSTANCE = FfiConverterOptionalIterationOrder{} diff --git a/bindings/go/uniffi/slatedb.h b/bindings/go/uniffi/slatedb.h index ed46683e11..ed9ac82954 100644 --- a/bindings/go/uniffi/slatedb.h +++ b/bindings/go/uniffi/slatedb.h @@ -1004,6 +1004,11 @@ uint64_t uniffi_slatedb_uniffi_fn_method_db_scan_with_options(uint64_t ptr, Rust uint64_t uniffi_slatedb_uniffi_fn_method_db_shutdown(uint64_t ptr ); #endif +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_DB_SHUTDOWN_WITH_OPTIONS +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_DB_SHUTDOWN_WITH_OPTIONS +uint64_t uniffi_slatedb_uniffi_fn_method_db_shutdown_with_options(uint64_t ptr, RustBuffer options +); +#endif #ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_DB_SNAPSHOT #define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_DB_SNAPSHOT uint64_t uniffi_slatedb_uniffi_fn_method_db_snapshot(uint64_t ptr @@ -2425,6 +2430,12 @@ uint16_t uniffi_slatedb_uniffi_checksum_method_db_scan_with_options(void #define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_DB_SHUTDOWN uint16_t uniffi_slatedb_uniffi_checksum_method_db_shutdown(void +); +#endif +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_DB_SHUTDOWN_WITH_OPTIONS +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_DB_SHUTDOWN_WITH_OPTIONS +uint16_t uniffi_slatedb_uniffi_checksum_method_db_shutdown_with_options(void + ); #endif #ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_DB_SNAPSHOT diff --git a/bindings/go/uniffi/slatedb_test.go b/bindings/go/uniffi/slatedb_test.go index 5facf2507c..79cff3ffb9 100644 --- a/bindings/go/uniffi/slatedb_test.go +++ b/bindings/go/uniffi/slatedb_test.go @@ -520,6 +520,38 @@ func TestDbLifecycleAndStatus(t *testing.T) { } } +func TestDbShutdownWithOptions(t *testing.T) { + wal := slatedb.FlushTypeWal + tests := []struct { + name string + flushType *slatedb.FlushType + }{ + {name: "wal", flushType: &wal}, + {name: "none", flushType: nil}, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + store := newMemoryStore(t) + handle := openTestDB(t, store, nil) + + if _, err := handle.db.Put([]byte("shutdown-options"), []byte("value")); err != nil { + t.Fatalf("Put(): %v", err) + } + + if err := handle.db.ShutdownWithOptions(slatedb.CloseOptions{FlushType: tt.flushType}); err != nil { + t.Fatalf("ShutdownWithOptions(): %v", err) + } + handle.open = false + + status := handle.db.Status() + if status.CloseReason == nil || *status.CloseReason != slatedb.CloseReasonClean { + t.Fatalf("Status() after ShutdownWithOptions(): got close reason %v, want %v", status.CloseReason, slatedb.CloseReasonClean) + } + }) + } +} + type fixedThreeByteSegmentExtractor struct{} func (fixedThreeByteSegmentExtractor) Name() string { return "fixed_three_byte" } diff --git a/bindings/uniffi/src/config.rs b/bindings/uniffi/src/config.rs index 6b5ddcdf43..a6209b5ee0 100644 --- a/bindings/uniffi/src/config.rs +++ b/bindings/uniffi/src/config.rs @@ -366,6 +366,30 @@ impl From for slatedb::config::FlushOptions { } } +/// Options controlling how a database is shut down. +#[derive(Clone, Debug, uniffi::Record)] +pub struct CloseOptions { + /// The final flush to perform before shutdown. When `None`, no final flush is + /// triggered and writes that are not durable may be lost. + pub flush_type: Option, +} + +impl Default for CloseOptions { + fn default() -> Self { + Self { + flush_type: Some(FlushType::MemTable), + } + } +} + +impl From for slatedb::config::CloseOptions { + fn from(value: CloseOptions) -> Self { + slatedb::config::CloseOptions { + flush_type: value.flush_type.map(Into::into), + } + } +} + /// Garbage collector options for one age-thresholded directory. #[derive(Clone, Debug, uniffi::Record)] pub struct GarbageCollectorDirectoryOptions { @@ -519,7 +543,37 @@ impl From for slatedb::config::GarbageCollectorOptions #[cfg(test)] mod tests { - use super::{GarbageCollectorOptions, ReaderOptions}; + use super::{CloseOptions, FlushType, GarbageCollectorOptions, ReaderOptions}; + + #[test] + fn close_options_default_flushes_memtable() { + let options: slatedb::config::CloseOptions = CloseOptions::default().into(); + + assert!(matches!( + options.flush_type, + Some(slatedb::config::FlushType::MemTable) + )); + } + + #[test] + fn close_options_can_flush_wal_only() { + let options: slatedb::config::CloseOptions = CloseOptions { + flush_type: Some(FlushType::Wal), + } + .into(); + + assert!(matches!( + options.flush_type, + Some(slatedb::config::FlushType::Wal) + )); + } + + #[test] + fn close_options_can_skip_final_flush() { + let options: slatedb::config::CloseOptions = CloseOptions { flush_type: None }.into(); + + assert!(options.flush_type.is_none()); + } #[test] fn boundary_files_are_enabled_by_default() { diff --git a/bindings/uniffi/src/db.rs b/bindings/uniffi/src/db.rs index 4e4dd22406..5ad2d6a635 100644 --- a/bindings/uniffi/src/db.rs +++ b/bindings/uniffi/src/db.rs @@ -1,7 +1,8 @@ use std::sync::Arc; use crate::config::{ - FlushOptions, IsolationLevel, MergeOptions, PutOptions, ReadOptions, ScanOptions, WriteOptions, + CloseOptions, FlushOptions, IsolationLevel, MergeOptions, PutOptions, ReadOptions, ScanOptions, + WriteOptions, }; use crate::db_snapshot::DbSnapshot; use crate::db_transaction::DbTransaction; @@ -43,6 +44,15 @@ impl Db { self.inner.close().await.map_err(Into::into) } + /// Performs the requested final flush and closes the database. + #[uniffi::method(name = "shutdown_with_options")] + pub async fn close_with_options(&self, options: CloseOptions) -> Result<(), Error> { + self.inner + .close_with_options(options.into()) + .await + .map_err(Into::into) + } + /// Reads the current value for `key`. pub async fn get(&self, key: Vec) -> Result>, Error> { validate_key(&key)?; diff --git a/bindings/uniffi/src/lib.rs b/bindings/uniffi/src/lib.rs index bf28ac7a4b..d287b19aeb 100644 --- a/bindings/uniffi/src/lib.rs +++ b/bindings/uniffi/src/lib.rs @@ -24,7 +24,7 @@ mod write_handle; pub use admin::Admin; pub use builder::{AdminBuilder, CloneBuilder, DbBuilder, DbReaderBuilder}; pub use config::{ - DurabilityLevel, FlushOptions, FlushType, GarbageCollectorDirectoryOptions, + CloseOptions, DurabilityLevel, FlushOptions, FlushType, GarbageCollectorDirectoryOptions, GarbageCollectorOptions, GarbageCollectorScheduleOptions, IsolationLevel, IterationOrder, MergeOptions, PutOptions, ReadOptions, ReaderMode, ReaderOptions, ScanOptions, SstBlockSize, Ttl, WriteOptions, diff --git a/slatedb/src/config.rs b/slatedb/src/config.rs index 2150f3ee08..5a2a9537ae 100644 --- a/slatedb/src/config.rs +++ b/slatedb/src/config.rs @@ -435,7 +435,7 @@ impl ScanOptions { } /// Enum representing the type of flush to perform. -#[derive(Clone)] +#[derive(Clone, Debug)] pub enum FlushType { /// Freeze the active memtable [crate::mem_table::KVTable] and write /// all immutable memtable entries (including the formerly active @@ -464,25 +464,28 @@ impl Default for FlushOptions { /// Options controlling how a database is closed. #[derive(Clone, Debug)] pub struct CloseOptions { - /// Whether to trigger a final flush of the active memtable before closing. - /// - /// When `false`, memtables already being flushed continue through the - /// existing shutdown pipeline. Defaults to `true`. - pub flush_memtables: bool, + /// The type of flush to perform before closing. + /// + /// When `None`, no final flush is triggered. Memtables already being + /// flushed continue through the existing shutdown pipeline, and writes + /// that are not durable may be lost. When set to `Some` flushes the + /// database in accordance with the specified [`FlushType`] + /// Defaults to `Some(FlushType::MemTable)`. + pub flush_type: Option, } impl Default for CloseOptions { fn default() -> Self { Self { - flush_memtables: true, + flush_type: Some(FlushType::MemTable), } } } impl CloseOptions { - /// Configure whether the active memtable is flushed before closing. - pub fn with_flush_memtables(mut self, flush_memtables: bool) -> Self { - self.flush_memtables = flush_memtables; + /// Configure the type of flush to perform before closing. + pub fn with_flush_type(mut self, flush_type: Option) -> Self { + self.flush_type = flush_type; self } } diff --git a/slatedb/src/db.rs b/slatedb/src/db.rs index bb2f9b2a12..17d75dad2d 100644 --- a/slatedb/src/db.rs +++ b/slatedb/src/db.rs @@ -686,36 +686,26 @@ impl Db { /// Close the database with custom options. /// - /// Setting [`CloseOptions::flush_memtables`] to `false` skips the final - /// active memtable flush. Memtables already being flushed are allowed to - /// finish, and writes that are not durable may be lost. pub async fn close_with_options(&self, options: CloseOptions) -> Result<(), crate::Error> { - let should_flush = match self.status().close_reason { + let flush_type = match self.status().close_reason { // If already closed, don't close again. Some(CloseReason::Clean) => return Err(SlateDBError::Closed.into()), // If in failed state, allow close, but don't flush since the database // might be in a bad state. Note that multiple close() calls will always // run when in a failed state (vs. a clean closure, which will return // Error::Closed(CloseReason::Clean) on subsequent calls). - Some(_) => false, + Some(_) => None, // Flush outstanding writes if the database is still open and the - // caller requested a final memtable flush. - None => options.flush_memtables, + // caller requested a final flush. + None => options.flush_type, }; // Mark the database as closed before flushing. self.inner.status_manager.write_result(Ok(())); - let result = if should_flush { - // Flush memtables to L0 so that the WAL does not need to be - // replayed on the next startup. + let result = if let Some(flush_type) = flush_type { self.inner - .flush( - FlushOptions { - flush_type: FlushType::MemTable, - }, - false, - ) + .flush(FlushOptions { flush_type }, false) .await .map_err(Into::into) .inspect_err(|e| warn!("failed to flush db during close [error={:?}]", e)) @@ -3511,6 +3501,46 @@ mod tests { ); } + #[tokio::test] + async fn test_close_with_options_flushes_wal_only() { + let object_store: Arc = Arc::new(InMemory::new()); + let mut settings = test_db_options(0, 1024, None); + settings.flush_interval = None; + let path = "/tmp/test_close_with_options_flushes_wal_only"; + let db = Db::builder(path, object_store.clone()) + .with_settings(settings.clone()) + .build() + .await + .unwrap(); + + db.put(b"test_key", b"test_value").await.unwrap(); + + db.close_with_options(CloseOptions::default().with_flush_type(Some(FlushType::Wal))) + .await + .unwrap(); + + let manifest = db.manifest(); + assert!( + manifest.manifest.core.replay_after_wal_id + 1 < manifest.manifest.core.next_wal_sst_id, + "expected a data WAL after the replay watermark" + ); + assert!( + manifest.manifest.core.tree.l0.is_empty(), + "expected no flushed memtables in the manifest" + ); + + let reopened = Db::builder(path, object_store) + .with_settings(settings) + .build() + .await + .unwrap(); + assert_eq!( + reopened.get(b"test_key").await.unwrap(), + Some(Bytes::from_static(b"test_value")) + ); + reopened.close().await.unwrap(); + } + #[tokio::test] async fn test_close_with_options_skips_final_flush() { let object_store: Arc = Arc::new(InMemory::new()); @@ -3529,10 +3559,15 @@ mod tests { db.put(b"test_key", b"test_value").await.unwrap(); - db.close_with_options(CloseOptions::default().with_flush_memtables(false)) + db.close_with_options(CloseOptions::default().with_flush_type(None)) .await .unwrap(); + assert_eq!( + lookup_metric(&metrics_recorder, crate::wal_buffer::stats::WAL_FLUSH_BYTES) + .unwrap_or(0), + 0 + ); assert_eq!( lookup_metric(&metrics_recorder, crate::db_stats::L0_FLUSH_BYTES).unwrap_or(0), 0 @@ -3557,7 +3592,7 @@ mod tests { let handle = db.put(b"key", b"value").await.unwrap(); - db.close_with_options(CloseOptions::default().with_flush_memtables(false)) + db.close_with_options(CloseOptions::default().with_flush_type(None)) .await .unwrap(); diff --git a/slatedb/src/db_reader.rs b/slatedb/src/db_reader.rs index ee98efb388..8b3b05a5aa 100644 --- a/slatedb/src/db_reader.rs +++ b/slatedb/src/db_reader.rs @@ -2827,13 +2827,8 @@ mod tests { // memtable flush does not flush the WAL, so an unawaited write would // race the flush interval and might never reach a WAL SST. let wal_only_key = b"wal_only_key"; - db.put(wal_only_key, b"wal_only_value") - .await - .unwrap() - .await_durable() - .await - .unwrap(); - db.close_with_options(CloseOptions::default().with_flush_memtables(false)) + db.put(wal_only_key, b"wal_only_value").await.unwrap(); + db.close_with_options(CloseOptions::default().with_flush_type(Some(FlushType::Wal))) .await .unwrap(); @@ -2918,7 +2913,7 @@ mod tests { .create_checkpoint(CheckpointScope::Durable, &CheckpointOptions::default()) .await .unwrap(); - db.close_with_options(CloseOptions::default().with_flush_memtables(false)) + db.close_with_options(CloseOptions::default().with_flush_type(Some(FlushType::Wal))) .await .unwrap(); From d12891f1a4ae7e835700330fa9adcecae3fb6c18 Mon Sep 17 00:00:00 2001 From: Chris Date: Wed, 5 Aug 2026 12:23:22 -0700 Subject: [PATCH 24/65] Replace nightly bencher work with benchmark.slatedb.io (#2009) --- .github/workflows/nightly.yaml | 151 -------------- slatedb-bencher/README.md | 34 ++- slatedb-bencher/benchmark-db.sh | 162 -------------- slatedb-bencher/benchmark-transaction.sh | 197 ------------------ .../docs/docs/operations/benchmarks.mdx | 53 ++++- 5 files changed, 70 insertions(+), 527 deletions(-) diff --git a/.github/workflows/nightly.yaml b/.github/workflows/nightly.yaml index 62b9ee9091..878eb63630 100644 --- a/.github/workflows/nightly.yaml +++ b/.github/workflows/nightly.yaml @@ -54,157 +54,6 @@ jobs: if: steps.update_microbenchmark_result.outcome == 'failure' run: exit 1 - benchmarks: - runs-on: warp-ubuntu-latest-x64-16x - timeout-minutes: 20 - steps: - - uses: actions/checkout@v4 - - - name: Install dependencies - run: | - sudo apt-get update - sudo apt-get install -y traceroute - sudo snap install aws-cli --classic - - - name: System information - env: - AWS_ACCESS_KEY_ID: ${{ secrets.TIGRIS_AWS_ACCESS_KEY_ID }} - AWS_SECRET_ACCESS_KEY: ${{ secrets.TIGRIS_AWS_SECRET_ACCESS_KEY }} - AWS_BUCKET: ${{ secrets.TIGRIS_AWS_BUCKET_BENCHER }} - AWS_REGION: auto - AWS_ENDPOINT: https://t3.storage.dev - run: | - echo "=== CPU ===" - lscpu - echo -e "\n=== Memory ===" - free -h - echo -e "\n=== Disk Space ===" - df -h - echo -e "\n=== Workspace Directory ===" - du -sh ${{ github.workspace }} - echo -e "\n=== Network ===" - traceroute t3.storage.dev - echo -e "Generating 1 gig file" - dd if=/dev/urandom of=/tmp/1gig bs=1G count=1 - echo -e "Uploading 1 gig file" - time aws s3 cp --endpoint-url $AWS_ENDPOINT /tmp/1gig s3://${{ secrets.TIGRIS_AWS_BUCKET_BENCHER }}/1gig - echo -e "Downloading 1 gig file" - time aws s3 cp --endpoint-url $AWS_ENDPOINT s3://${{ secrets.TIGRIS_AWS_BUCKET_BENCHER }}/1gig /tmp/1gig - echo -e "Deleting 1 gig file" - time aws s3 rm --endpoint-url $AWS_ENDPOINT s3://${{ secrets.TIGRIS_AWS_BUCKET_BENCHER }}/1gig - - - name: Download previous mermaid plots - uses: actions/cache/restore@v4 - with: - path: ./target/bencher/results/mermaid - key: ${{ runner.os }}-mermaid-plots-${{ github.run_id }} - restore-keys: | - ${{ runner.os }}-mermaid-plots- - - - name: Run benchmark - env: - CLOUD_PROVIDER: aws - AWS_ACCESS_KEY_ID: ${{ secrets.TIGRIS_AWS_ACCESS_KEY_ID }} - AWS_SECRET_ACCESS_KEY: ${{ secrets.TIGRIS_AWS_SECRET_ACCESS_KEY }} - AWS_BUCKET: ${{ secrets.TIGRIS_AWS_BUCKET_BENCHER }} - AWS_REGION: auto - AWS_ENDPOINT: https://t3.storage.dev - SLATEDB_BENCH_CLEAN: true - RUST_LOG: info - run: | - aws s3 rm --endpoint-url $AWS_ENDPOINT s3://${{ secrets.TIGRIS_AWS_BUCKET_BENCHER }} --recursive - ./slatedb-bencher/benchmark-db.sh - - - name: Save mermaid plots cache - uses: actions/cache/save@v4 - with: - path: ./target/bencher/results/mermaid - key: ${{ runner.os }}-mermaid-plots-${{ github.run_id }} - - - name: Add mermaid diagrams to summary - run: | - echo "# SlateDB Benchmark Results" >> $GITHUB_STEP_SUMMARY - echo "" >> $GITHUB_STEP_SUMMARY - - # Add each mermaid diagram to the summary - for mermaid_file in target/bencher/results/mermaid/*.mermaid; do - if [ -f "$mermaid_file" ]; then - echo "" >> $GITHUB_STEP_SUMMARY - echo '```mermaid' >> $GITHUB_STEP_SUMMARY - cat "$mermaid_file" >> $GITHUB_STEP_SUMMARY - echo '```' >> $GITHUB_STEP_SUMMARY - echo "" >> $GITHUB_STEP_SUMMARY - fi - done - - echo "Mermaid diagrams added to GitHub Actions summary!" - echo "Total diagrams: $(ls -1 target/bencher/results/mermaid/*.mermaid 2>/dev/null | wc -l)" - - transaction-benchmarks: - runs-on: warp-ubuntu-latest-x64-16x - # Must run after `benchmarks` because both jobs clear the same bucket prefix; - # running concurrently can delete each other's manifest/object files mid-run. - # This also reduces noise between the two tests since the bucket's resources - # (network, disk, etc) should be used by only one test at a time, giving more - # stable results. - needs: benchmarks - timeout-minutes: 30 - steps: - - uses: actions/checkout@v4 - - - name: Install dependencies - run: | - sudo apt-get update - sudo apt-get install -y traceroute - sudo snap install aws-cli --classic - - - name: Download previous transaction mermaid plots - uses: actions/cache/restore@v4 - with: - path: ./target/bencher/transaction-results/mermaid - key: ${{ runner.os }}-txn-mermaid-plots-${{ github.run_id }} - restore-keys: | - ${{ runner.os }}-txn-mermaid-plots- - - - name: Run transaction benchmark - env: - CLOUD_PROVIDER: aws - AWS_ACCESS_KEY_ID: ${{ secrets.TIGRIS_AWS_ACCESS_KEY_ID }} - AWS_SECRET_ACCESS_KEY: ${{ secrets.TIGRIS_AWS_SECRET_ACCESS_KEY }} - AWS_BUCKET: ${{ secrets.TIGRIS_AWS_BUCKET_BENCHER }} - AWS_REGION: auto - AWS_ENDPOINT: https://t3.storage.dev - SLATEDB_BENCH_CLEAN: true - RUST_LOG: info - run: | - aws s3 rm --endpoint-url $AWS_ENDPOINT s3://${{ secrets.TIGRIS_AWS_BUCKET_BENCHER }} --recursive - ./slatedb-bencher/benchmark-transaction.sh - - - name: Save transaction mermaid plots cache - uses: actions/cache/save@v4 - with: - path: ./target/bencher/transaction-results/mermaid - key: ${{ runner.os }}-txn-mermaid-plots-${{ github.run_id }} - - - name: Add transaction mermaid diagrams to summary - run: | - echo "# SlateDB Transaction Benchmark Results" >> $GITHUB_STEP_SUMMARY - echo "" >> $GITHUB_STEP_SUMMARY - - # Add each mermaid diagram to the summary - for mermaid_file in target/bencher/transaction-results/mermaid/*.mermaid; do - if [ -f "$mermaid_file" ]; then - echo "" >> $GITHUB_STEP_SUMMARY - echo '```mermaid' >> $GITHUB_STEP_SUMMARY - cat "$mermaid_file" >> $GITHUB_STEP_SUMMARY - echo '```' >> $GITHUB_STEP_SUMMARY - echo "" >> $GITHUB_STEP_SUMMARY - fi - done - - echo "Transaction mermaid diagrams added to GitHub Actions summary!" - echo "Total diagrams: $(ls -1 target/bencher/transaction-results/mermaid/*.mermaid 2>/dev/null | wc -l)" - microbenchmark-pprofs: runs-on: warp-ubuntu-latest-arm64-8x steps: diff --git a/slatedb-bencher/README.md b/slatedb-bencher/README.md index 7aca293c36..4c8ba6f97f 100644 --- a/slatedb-bencher/README.md +++ b/slatedb-bencher/README.md @@ -58,10 +58,10 @@ following environment variables before benchmarking: ## `benchmark-db.sh` -There is also a shell script which runs a series of benchmarks and then draws -the plots using `gnuplot`. Think of it as a template to start with to create -a set of benchmarks suitable for your task. The script should be run from -the repository root: +There is also a shell script which runs a series of benchmarks and records +the results. Think of it as a template to start with to create a set of +benchmarks suitable for your task. The script should be run from the repository +root: ```bash ./slatedb-bencher/benchmark-db.sh @@ -69,16 +69,32 @@ the repository root: The command above will produce results at `target/bencher/results` directory. The results include: -- `plots`: Plots for each benchmark - `dats`: Data files for each benchmark - `logs`: Log files for each benchmark -- `benchmark-data.json`: A JSON file containing all the benchmark results in [github-action-benchmark](https://github.com/benchmark-action/github-action-benchmark) format. -The script also has a `SLATEDB_BENCH_CLEAN` environment variable which can be set to `true` to clean up the test data in object storage after each benchmark. +### Plotting results with `gnuplot` + +The `.dat` files are whitespace-delimited, with columns for elapsed time, +puts per second, and gets per second. After installing `gnuplot`, you can render +a result file to a PNG with: + +```bash +gnuplot <<'EOF' +set terminal pngcairo size 1280,720 +set output "target/bencher/results/20_1.png" +set title "SlateDB benchmark: 20% puts, concurrency 1" +set xlabel "Elapsed time (seconds)" +set ylabel "Requests per second" +set key outside +plot "target/bencher/results/dats/20_1.dat" using 1:2 with lines title "puts/s", \ + "target/bencher/results/dats/20_1.dat" using 1:3 with lines title "gets/s" +EOF +``` -### `nightly.yaml` +Replace `20_1.dat` and the labels with the benchmark configuration you want to +plot. -`benchmark-db.sh` is also used in `.github/workflows/nightly.yaml` to benchmark the nightly build. The tests are run using [WarpBuild](https://warpbuild.com), and each run appends to mermaid `xyChart` files that are posted to the workflow's GitHub Actions job summary. +The script also has a `SLATEDB_BENCH_CLEAN` environment variable which can be set to `true` to clean up the test data in object storage after each benchmark. ## `compaction` Subcommand diff --git a/slatedb-bencher/benchmark-db.sh b/slatedb-bencher/benchmark-db.sh index 9ea2d3a7d0..f4d1978490 100755 --- a/slatedb-bencher/benchmark-db.sh +++ b/slatedb-bencher/benchmark-db.sh @@ -8,7 +8,6 @@ OUT="target/bencher/results" mkdir -p $OUT/dats mkdir -p $OUT/logs -mkdir -p $OUT/mermaid run_bench() { local put_percentage="$1" @@ -46,165 +45,6 @@ generate_dat() { grep "stats dump" "$input_file" | sed -E 's/.*elapsed ([0-9.]+).*put\/s: ([0-9.]+).*get\/s: ([0-9.]+).*/\1 \2 \3/' > "$output_file" } -generate_mermaid () { - local dat_file="$1" - local mermaid_file="$2" - - # Create mermaid directory if it doesn't exist - mkdir -p "$(dirname "$mermaid_file")" - - # Get the last line from dat file (most recent benchmark result) - if [ ! -f "$dat_file" ] || [ ! -s "$dat_file" ]; then - echo "Warning: dat file $dat_file does not exist or is empty" - return 1 - fi - - local last_line=$(tail -n 1 "$dat_file") - local put_value=$(echo "$last_line" | awk '{print $2}') - local get_value=$(echo "$last_line" | awk '{print $3}') - - # Get git commit hash (first 7 characters) - local git_hash=$(git rev-parse --short=7 HEAD 2>/dev/null || echo "unknown") - - # Get current date in YYYY-MM-DD format - local current_date=$(date +"%Y-%m-%d") - - # Create x-axis entry - local x_entry="$current_date ($git_hash)" - - # Extract put_percentage and concurrency from mermaid filename - local filename=$(basename "$mermaid_file" .mermaid) - local put_percentage=$(echo "$filename" | cut -d'_' -f1) - local concurrency=$(echo "$filename" | cut -d'_' -f2) - - # Calculate max value for y-axis scaling - local max_value=$(echo "$put_value $get_value" | tr ' ' '\n' | sort -nr | head -n1) - local y_max=$(echo "$max_value * 1.2" | bc -l | cut -d'.' -f1) - - if [ ! -f "$mermaid_file" ]; then - # Create new mermaid file - cat > "$mermaid_file" << EOF ---- -config: - xyChart: - chartOrientation: horizontal - height: 768 - width: 1024 - themeVariables: - xyChart: - plotColorPalette: '#1e81b0, #e28743' ---- -xychart-beta - title "SlateDB [puts=${put_percentage}%, threads=${concurrency}, 🔵=puts, 🟠=get]" - x-axis ["$x_entry"] - y-axis "requests/s" 0 --> $y_max - line [$put_value] - line [$get_value] -EOF - else - # Update existing mermaid file - local temp_file=$(mktemp) - - # Read current content (match only Mermaid series lines, not YAML like plotColorPalette) - local title_line=$(grep -E "^[[:space:]]*title[[:space:]]" "$mermaid_file" | sed 's/^[[:space:]]*//') - local x_axis_line=$(grep -E "^[[:space:]]*x-axis[[:space:]]*\\[" "$mermaid_file") - local put_line=$(grep -E "^[[:space:]]*line[[:space:]]*\\[" "$mermaid_file" | head -n1) - local get_line=$(grep -E "^[[:space:]]*line[[:space:]]*\\[" "$mermaid_file" | tail -n1) - - # Extract current values - local current_x_values=$(echo "$x_axis_line" | sed 's/.*\[//;s/\].*//' | tr ',' '\n' | sed 's/^[[:space:]]*"//;s/"[[:space:]]*$//') - local current_put_values=$(echo "$put_line" | sed 's/.*\[//;s/\].*//') - local current_get_values=$(echo "$get_line" | sed 's/.*\[//;s/\].*//') - - # Convert to arrays - local x_array=() - local put_array=() - local get_array=() - - # Parse existing x-axis values - while IFS= read -r line; do - if [ -n "$line" ]; then - x_array+=("$line") - fi - done <<< "$current_x_values" - - # Parse existing put values - IFS=',' read -ra put_array <<< "$current_put_values" - - # Parse existing get values - IFS=',' read -ra get_array <<< "$current_get_values" - - # Prepend new values (newest first) - x_array=("$x_entry" "${x_array[@]}") - put_array=("$put_value" "${put_array[@]}") - get_array=("$get_value" "${get_array[@]}") - - # Keep only first 30 values (newest-first) if we have more - if [ ${#x_array[@]} -gt 30 ]; then - x_array=("${x_array[@]:0:30}") - put_array=("${put_array[@]:0:30}") - get_array=("${get_array[@]:0:30}") - fi - - # Build new x-axis string - local new_x_axis="x-axis [" - for i in "${!x_array[@]}"; do - if [ $i -gt 0 ]; then - new_x_axis="$new_x_axis, " - fi - new_x_axis="$new_x_axis\"${x_array[i]}\"" - done - new_x_axis="$new_x_axis]" - - # Build new put line string - local new_put_line="line [" - for i in "${!put_array[@]}"; do - if [ $i -gt 0 ]; then - new_put_line="$new_put_line, " - fi - new_put_line="$new_put_line${put_array[i]}" - done - new_put_line="$new_put_line]" - - # Build new get line string - local new_get_line="line [" - for i in "${!get_array[@]}"; do - if [ $i -gt 0 ]; then - new_get_line="$new_get_line, " - fi - new_get_line="$new_get_line${get_array[i]}" - done - new_get_line="$new_get_line]" - - # Calculate max value for y-axis scaling from all values - local all_values="${put_array[*]} ${get_array[*]}" - local max_value=$(echo "$all_values" | tr ' ' '\n' | sort -nr | head -n1) - local y_max=$(echo "$max_value * 1.2" | bc -l | cut -d'.' -f1) - - # Write updated mermaid file - cat > "$mermaid_file" << EOF ---- -config: - xyChart: - chartOrientation: horizontal - height: 768 - width: 1024 - themeVariables: - xyChart: - plotColorPalette: '#1e81b0, #e28743' ---- -xychart-beta - $title_line - $new_x_axis - y-axis "requests/s" 0 --> $y_max - $new_put_line - $new_get_line -EOF - fi - - echo "Generated/updated mermaid chart: $mermaid_file" -} - # Set CLOUD_PROVIDER to local if not already set export CLOUD_PROVIDER=${CLOUD_PROVIDER:-local} echo "Using cloud provider: $CLOUD_PROVIDER" @@ -220,11 +60,9 @@ for put_percentage in 20 40 60 80 100; do for concurrency in 1 32; do log_file="$OUT/logs/${put_percentage}_${concurrency}.log" dat_file="$OUT/dats/${put_percentage}_${concurrency}.dat" - mermaid_file="$OUT/mermaid/${put_percentage}_${concurrency}.mermaid" num_keys=$((put_percentage * 1000)) run_bench "$put_percentage" "$concurrency" "$num_keys" "$log_file" generate_dat "$log_file" "$dat_file" - generate_mermaid "$dat_file" "$mermaid_file" done done diff --git a/slatedb-bencher/benchmark-transaction.sh b/slatedb-bencher/benchmark-transaction.sh index 3f0e25f71b..55d6cc6311 100755 --- a/slatedb-bencher/benchmark-transaction.sh +++ b/slatedb-bencher/benchmark-transaction.sh @@ -7,7 +7,6 @@ OUT="target/bencher/transaction-results" mkdir -p "$OUT/logs" mkdir -p "$OUT/dats" -mkdir -p "$OUT/mermaid" # Define DB path once for both bencher and cleanup DB_PATH_NAME="slatedb-txn-bencher" @@ -72,195 +71,6 @@ generate_dat() { fi } -generate_mermaid() { - local dat_file="$1" - local mermaid_file="$2" - - # Create mermaid directory if it doesn't exist - mkdir -p "$(dirname "$mermaid_file")" - - # Get the last line from dat file (most recent benchmark result) - if [ ! -f "$dat_file" ] || [ ! -s "$dat_file" ]; then - echo "Warning: dat file $dat_file does not exist or is empty" - return 1 - fi - - local last_line=$(tail -n 1 "$dat_file") - local commit_value=$(echo "$last_line" | awk '{print $2}') - local abort_value=$(echo "$last_line" | awk '{print $3}') - local conflict_value=$(echo "$last_line" | awk '{print $4}') - local ops_value=$(echo "$last_line" | awk '{print $5}') - - # Get git commit hash (first 7 characters) - local git_hash=$(git rev-parse --short=7 HEAD 2>/dev/null || echo "unknown") - - # Get current date in YYYY-MM-DD format - local current_date=$(date +"%Y-%m-%d") - - # Create x-axis entry - local x_entry="$current_date ($git_hash)" - - # Extract test parameters from mermaid filename - # Format: isolation_concurrency_txnsize_mode.mermaid (e.g., snapshot_4_10_txn.mermaid) - local filename=$(basename "$mermaid_file" .mermaid) - local isolation=$(echo "$filename" | cut -d'_' -f1) - local concurrency=$(echo "$filename" | cut -d'_' -f2) - local txn_size=$(echo "$filename" | cut -d'_' -f3) - local mode=$(echo "$filename" | cut -d'_' -f4) - - local title_mode="Transaction" - if [ "$mode" = "batch" ]; then - title_mode="WriteBatch" - fi - - # Calculate max value for y-axis scaling (consider all 4 metrics) - local max_value=$(printf "%s\n" "$commit_value" "$abort_value" "$conflict_value" "$ops_value" | sort -nr | head -n1) - local y_max=$(awk -v m="$max_value" 'BEGIN{printf "%d", (m*1.2)}') - - if [ ! -f "$mermaid_file" ]; then - # Create new mermaid file - cat > "$mermaid_file" << EOF ---- -config: - xyChart: - chartOrientation: horizontal - height: 768 - width: 1024 - themeVariables: - xyChart: - plotColorPalette: '#2ecc71, #e74c3c, #f39c12, #3498db' ---- -xychart-beta - title "SlateDB Txn [${isolation}, threads=${concurrency}, txn_size=${txn_size}, ${title_mode}] 🟢=commit 🔴=abort 🟠=conflict 🔵=ops" - x-axis ["$x_entry"] - y-axis "requests/s" 0 --> $y_max - line [$commit_value] - line [$abort_value] - line [$conflict_value] - line [$ops_value] -EOF - else - # Update existing mermaid file - # Read current content (match only Mermaid series lines, not YAML like plotColorPalette) - local title_line=$(grep -E "^[[:space:]]*title[[:space:]]" "$mermaid_file" | sed 's/^[[:space:]]*//') - local x_axis_line=$(grep -E "^[[:space:]]*x-axis[[:space:]]*\\[" "$mermaid_file") - local commit_line=$(grep -E "^[[:space:]]*line[[:space:]]*\\[" "$mermaid_file" | sed -n '1p') - local abort_line=$(grep -E "^[[:space:]]*line[[:space:]]*\\[" "$mermaid_file" | sed -n '2p') - local conflict_line=$(grep -E "^[[:space:]]*line[[:space:]]*\\[" "$mermaid_file" | sed -n '3p') - local ops_line=$(grep -E "^[[:space:]]*line[[:space:]]*\\[" "$mermaid_file" | sed -n '4p') - - # Extract current values - local current_x_values=$(echo "$x_axis_line" | sed 's/.*\[//;s/\].*//' | tr ',' '\n' | sed 's/^[[:space:]]*"//;s/"[[:space:]]*$//') - local current_commit_values=$(echo "$commit_line" | sed 's/.*\[//;s/\].*//') - local current_abort_values=$(echo "$abort_line" | sed 's/.*\[//;s/\].*//') - local current_conflict_values=$(echo "$conflict_line" | sed 's/.*\[//;s/\].*//') - local current_ops_values=$(echo "$ops_line" | sed 's/.*\[//;s/\].*//') - - # Convert to arrays - local x_array=() - local commit_array=() - local abort_array=() - local conflict_array=() - local ops_array=() - - # Parse existing x-axis values - while IFS= read -r line; do - if [ -n "$line" ]; then - x_array+=("$line") - fi - done <<< "$current_x_values" - - # Parse existing values - IFS=',' read -ra commit_array <<< "$current_commit_values" - IFS=',' read -ra abort_array <<< "$current_abort_values" - IFS=',' read -ra conflict_array <<< "$current_conflict_values" - IFS=',' read -ra ops_array <<< "$current_ops_values" - - # Trim whitespace from array elements (comma-split can leave leading spaces) - # Use separate loops to handle potential length mismatches safely - for i in "${!commit_array[@]}"; do commit_array[i]="${commit_array[i]//[[:space:]]/}"; done - for i in "${!abort_array[@]}"; do abort_array[i]="${abort_array[i]//[[:space:]]/}"; done - for i in "${!conflict_array[@]}"; do conflict_array[i]="${conflict_array[i]//[[:space:]]/}"; done - for i in "${!ops_array[@]}"; do ops_array[i]="${ops_array[i]//[[:space:]]/}"; done - - # Prepend new values (newest first) - x_array=("$x_entry" "${x_array[@]}") - commit_array=("$commit_value" "${commit_array[@]}") - abort_array=("$abort_value" "${abort_array[@]}") - conflict_array=("$conflict_value" "${conflict_array[@]}") - ops_array=("$ops_value" "${ops_array[@]}") - - # Keep only first 30 values (newest-first) if we have more - if [ ${#x_array[@]} -gt 30 ]; then - x_array=("${x_array[@]:0:30}") - commit_array=("${commit_array[@]:0:30}") - abort_array=("${abort_array[@]:0:30}") - conflict_array=("${conflict_array[@]:0:30}") - ops_array=("${ops_array[@]:0:30}") - fi - - # Build new x-axis string - local new_x_axis="x-axis [" - for i in "${!x_array[@]}"; do - if [ $i -gt 0 ]; then - new_x_axis="$new_x_axis, " - fi - new_x_axis="$new_x_axis\"${x_array[i]}\"" - done - new_x_axis="$new_x_axis]" - - # Build new line strings - local new_commit_line="line [" - local new_abort_line="line [" - local new_conflict_line="line [" - local new_ops_line="line [" - for i in "${!commit_array[@]}"; do - if [ $i -gt 0 ]; then - new_commit_line="$new_commit_line, " - new_abort_line="$new_abort_line, " - new_conflict_line="$new_conflict_line, " - new_ops_line="$new_ops_line, " - fi - new_commit_line="$new_commit_line${commit_array[i]}" - new_abort_line="$new_abort_line${abort_array[i]}" - new_conflict_line="$new_conflict_line${conflict_array[i]}" - new_ops_line="$new_ops_line${ops_array[i]}" - done - new_commit_line="$new_commit_line]" - new_abort_line="$new_abort_line]" - new_conflict_line="$new_conflict_line]" - new_ops_line="$new_ops_line]" - - # Calculate max value for y-axis scaling from all values (all 4 metrics) - local max_value=$(printf "%s\n" "${commit_array[@]}" "${abort_array[@]}" "${conflict_array[@]}" "${ops_array[@]}" | sort -nr | head -n1) - local y_max=$(awk -v m="$max_value" 'BEGIN{printf "%d", (m*1.2)}') - - # Write updated mermaid file - cat > "$mermaid_file" << EOF ---- -config: - xyChart: - chartOrientation: horizontal - height: 768 - width: 1024 - themeVariables: - xyChart: - plotColorPalette: '#2ecc71, #e74c3c, #f39c12, #3498db' ---- -xychart-beta - $title_line - $new_x_axis - y-axis "requests/s" 0 --> $y_max - $new_commit_line - $new_abort_line - $new_conflict_line - $new_ops_line -EOF - fi - - echo "Generated/updated mermaid chart: $mermaid_file" -} - # Set CLOUD_PROVIDER to local if not already set export CLOUD_PROVIDER=${CLOUD_PROVIDER:-local} echo "Using cloud provider: $CLOUD_PROVIDER" @@ -283,41 +93,34 @@ echo "" echo "Test 1: Low concurrency with Transactions (Snapshot)" run_txn_bench "snapshot" 4 10 false "$OUT/logs/snapshot_4_10_txn.log" generate_dat "$OUT/logs/snapshot_4_10_txn.log" "$OUT/dats/snapshot_4_10_txn.dat" -generate_mermaid "$OUT/dats/snapshot_4_10_txn.dat" "$OUT/mermaid/snapshot_4_10_txn.mermaid" # Test 2: Low concurrency, Snapshot isolation, WriteBatch echo "Test 2: Low concurrency with WriteBatch" run_txn_bench "snapshot" 4 10 true "$OUT/logs/snapshot_4_10_batch.log" generate_dat "$OUT/logs/snapshot_4_10_batch.log" "$OUT/dats/snapshot_4_10_batch.dat" -generate_mermaid "$OUT/dats/snapshot_4_10_batch.dat" "$OUT/mermaid/snapshot_4_10_batch.mermaid" # Test 3: High concurrency, Snapshot isolation, Transaction echo "Test 3: High concurrency with Transactions (Snapshot)" run_txn_bench "snapshot" 16 10 false "$OUT/logs/snapshot_16_10_txn.log" generate_dat "$OUT/logs/snapshot_16_10_txn.log" "$OUT/dats/snapshot_16_10_txn.dat" -generate_mermaid "$OUT/dats/snapshot_16_10_txn.dat" "$OUT/mermaid/snapshot_16_10_txn.mermaid" # Test 4: High concurrency, Snapshot isolation, WriteBatch echo "Test 4: High concurrency with WriteBatch" run_txn_bench "snapshot" 16 10 true "$OUT/logs/snapshot_16_10_batch.log" generate_dat "$OUT/logs/snapshot_16_10_batch.log" "$OUT/dats/snapshot_16_10_batch.dat" -generate_mermaid "$OUT/dats/snapshot_16_10_batch.dat" "$OUT/mermaid/snapshot_16_10_batch.mermaid" # Test 5: High concurrency, SerializableSnapshot isolation echo "Test 5: High concurrency with SerializableSnapshot" run_txn_bench "serializable" 16 10 false "$OUT/logs/serializable_16_10_txn.log" generate_dat "$OUT/logs/serializable_16_10_txn.log" "$OUT/dats/serializable_16_10_txn.dat" -generate_mermaid "$OUT/dats/serializable_16_10_txn.dat" "$OUT/mermaid/serializable_16_10_txn.mermaid" # Test 6: Large transactions echo "Test 6: Large transactions (50 ops)" run_txn_bench "snapshot" 8 50 false "$OUT/logs/snapshot_8_50_txn.log" generate_dat "$OUT/logs/snapshot_8_50_txn.log" "$OUT/dats/snapshot_8_50_txn.dat" -generate_mermaid "$OUT/dats/snapshot_8_50_txn.dat" "$OUT/mermaid/snapshot_8_50_txn.mermaid" echo "" echo "=== Benchmark Complete ===" echo "Results saved to:" echo " Logs: $OUT/logs/" echo " Data: $OUT/dats/" -echo " Charts: $OUT/mermaid/" diff --git a/website/src/content/docs/docs/operations/benchmarks.mdx b/website/src/content/docs/docs/operations/benchmarks.mdx index 4f8eb84f77..5f46077ad1 100644 --- a/website/src/content/docs/docs/operations/benchmarks.mdx +++ b/website/src/content/docs/docs/operations/benchmarks.mdx @@ -1,18 +1,55 @@ --- title: Benchmarks -description: See nightly performance results and learn how to run your own benchmarks +description: Review SlateDB performance results and learn how to run your own benchmarks --- -SlateDB currently has two benchmarking tools: **bencher** and **microbenchmarks**. +## Release benchmarks -- [bencher](https://github.com/slatedb/slatedb/tree/main/slatedb-bencher) is a tool to benchmark put/get operations against an object store. You can configure the tool with a variety of options. See [bencher/README.md](https://github.com/slatedb/slatedb/blob/main/slatedb-bencher/README.md) for details. [benchmark-db.sh](https://github.com/slatedb/slatedb/blob/main/slatedb-bencher/benchmark-db.sh) also serves as an example. +SlateDB publishes release benchmark results at +[benchmark.slatedb.io](https://benchmark.slatedb.io). The +[slatedb/slatedb-benchmark](https://github.com/slatedb/slatedb-benchmark) +repository contains the runner and workload definitions. It also stores the +raw results published on the site. -- Microbenchmarks run with [Criterion](https://bheisler.github.io/criterion.rs/). They are located in [benches](https://github.com/slatedb/slatedb/tree/main/slatedb/benches) and run for specific internal SlateDB functions. A comment is left on all PRs when a > 200% slowdown is detected. +The release suite starts from a shared database with 300 million records, about +120 GiB of logical data. The runner applies a fixed workload catalog to each +SlateDB revision. The catalog combines relevant workloads from +[YCSB Core](https://github.com/brianfrankcooper/YCSB/wiki/Core-Workloads) and +RocksDB's +[`db_bench`](https://github.com/facebook/rocksdb/wiki/Benchmarking-tools). +SlateDB-specific cases exercise idle behavior and transaction contention. Most +workloads run 64 closed-loop clients after a five-minute warmup and record 15 +minutes of activity. -### Nightly Benchmarks +## Other benchmarking tools -We run both **bencher** and **microbenchmarks** nightly. The [results](https://github.com/slatedb/slatedb/actions/workflows/nightly.yaml) are published in the Github action summary. +- [slatedb-bencher](https://github.com/slatedb/slatedb/tree/main/slatedb-bencher) + runs configurable database, compaction, and transaction benchmarks against + an object store. Its + [README](https://github.com/slatedb/slatedb/blob/main/slatedb-bencher/README.md) + documents the command-line options, and + [benchmark-db.sh](https://github.com/slatedb/slatedb/blob/main/slatedb-bencher/benchmark-db.sh) + provides an example workload matrix. -- **Bencher** benchmarks run on [WarpBuild](https://warpbuild.com)'s [warp-ubuntu-latest-x64-16x](https://docs.warpbuild.com/cloud-runners) runners, which use [Hetzner](https://hetzner.com/) machines in Frankfurt. We use [Tigris](https://www.tigrisdata.com/) for object storage with the `auto` region setting, which resolves to Frankfurt as well. Bandwidth between WarpBuild (Hetzner) and Tigris seems to be about 500MiB/s down and 130MiB/s up. We routinely max out the bandwidth in the nightly tests. +- SlateDB uses [Criterion](https://bheisler.github.io/criterion.rs/) for + microbenchmarks of internal functions. The benchmark sources live in + [slatedb/benches](https://github.com/slatedb/slatedb/tree/main/slatedb/benches). -- **Microbenchmarks** run on [standard Linux Github action runners](https://docs.github.com/en/actions/using-github-hosted-runners/using-github-hosted-runners/about-github-hosted-runners#standard-github-hosted-runners-for-public-repositories) with the [pprof-rs](https://github.com/tikv/pprof-rs) profiler. The resulting profiler protobuf files are published to [pprof.me](https://pprof.me) and links to each microbenchmark are provided in the Github action summary. +## Nightly microbenchmarks + +The +[nightly workflow](https://github.com/slatedb/slatedb/actions/workflows/nightly.yaml) +runs the Criterion microbenchmarks on standard Linux GitHub-hosted runners. It +also records profiles with [pprof-rs](https://github.com/tikv/pprof-rs) and +uploads them to [pprof.me](https://pprof.me). The GitHub Actions job summary +links to each profile. + +## Benchmarking object stores + +SlateDB benchmarks measure the database and object store together. Use +[MinIO Warp](https://github.com/minio/warp) to measure raw S3-compatible object +store performance without SlateDB in the request path. Warp runs concurrent +GET, PUT, DELETE, and mixed-request benchmarks with configurable object sizes +and concurrency. Run it from the same region and network used by your SlateDB +clients, ideally on the same machine, so the network path and client capacity +remain comparable. From 5b2a90545c095e69fb6f49ac47339c084b629760 Mon Sep 17 00:00:00 2001 From: Chris Date: Sun, 9 Aug 2026 10:50:28 -0700 Subject: [PATCH 25/65] Run nightly microbenchmarks on Warp ARM (#2016) --- .github/workflows/nightly.yaml | 2 +- website/src/content/docs/docs/operations/benchmarks.mdx | 7 ++++--- 2 files changed, 5 insertions(+), 4 deletions(-) diff --git a/.github/workflows/nightly.yaml b/.github/workflows/nightly.yaml index 878eb63630..24639cfb13 100644 --- a/.github/workflows/nightly.yaml +++ b/.github/workflows/nightly.yaml @@ -15,7 +15,7 @@ permissions: jobs: # Run and save nightly microbenchmark data so PRs have a fresh baseline to compare against microbenchmarks: - runs-on: ubuntu-latest + runs-on: warp-ubuntu-latest-arm64-8x steps: - uses: actions/checkout@v4 diff --git a/website/src/content/docs/docs/operations/benchmarks.mdx b/website/src/content/docs/docs/operations/benchmarks.mdx index 5f46077ad1..adfb7fc72c 100644 --- a/website/src/content/docs/docs/operations/benchmarks.mdx +++ b/website/src/content/docs/docs/operations/benchmarks.mdx @@ -39,9 +39,10 @@ minutes of activity. The [nightly workflow](https://github.com/slatedb/slatedb/actions/workflows/nightly.yaml) -runs the Criterion microbenchmarks on standard Linux GitHub-hosted runners. It -also records profiles with [pprof-rs](https://github.com/tikv/pprof-rs) and -uploads them to [pprof.me](https://pprof.me). The GitHub Actions job summary +runs the Criterion microbenchmarks on [WarpBuild](https://warpbuild.com)'s +[warp-ubuntu-latest-arm64-8x](https://docs.warpbuild.com/cloud-runners) ARM +runners. It also records profiles with [pprof-rs](https://github.com/tikv/pprof-rs) +and uploads them to [pprof.me](https://pprof.me). The GitHub Actions job summary links to each profile. ## Benchmarking object stores From bbb09be049d7362070524a439ba8c9fa8ca3619b Mon Sep 17 00:00:00 2001 From: Chris Date: Mon, 10 Aug 2026 07:36:53 -0700 Subject: [PATCH 26/65] Select previous release tag dynamically (#2017) --- .github/workflows/release.yaml | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/.github/workflows/release.yaml b/.github/workflows/release.yaml index 6eec0a7e3e..96765b5c65 100644 --- a/.github/workflows/release.yaml +++ b/.github/workflows/release.yaml @@ -32,6 +32,18 @@ jobs: with: ssh-key: ${{ secrets.RELEASE_SSH_KEY }} ref: ${{ github.ref }} + fetch-depth: 0 + + # Select the most recent stable release tag reachable from the release branch. + - name: Find previous release tag + id: previous_release + run: | + previous_tag="$(git describe \ + --tags \ + --abbrev=0 \ + --match 'v[0-9]*.[0-9]*.[0-9]*' \ + --exclude '*-*')" + echo "tag=${previous_tag}" >> "${GITHUB_OUTPUT}" # Set up Rust stable for installing cargo-edit - name: Setup Rust stable @@ -67,6 +79,7 @@ jobs: tag: v${{ github.event.inputs.version }} name: v${{ github.event.inputs.version }} generateReleaseNotes: true + generateReleaseNotesPreviousTag: ${{ steps.previous_release.outputs.tag }} token: ${{ secrets.GITHUB_TOKEN }} # Publish crate chain to crates.io in dependency order From 47040a37e1e44da66276a803cf6a6bd3526c7648 Mon Sep 17 00:00:00 2001 From: Rohan Date: Tue, 11 Aug 2026 02:31:04 -0400 Subject: [PATCH 27/65] [rfc-30 5/N]: add WalAdmin trait for administrative wal operations (#2014) --- rfcs/0030-pluggable-wal.md | 69 ++- slatedb/src/clone.rs | 567 ++++++++++++++++++------ slatedb/src/db/builder.rs | 11 +- slatedb/src/garbage_collector/wal_gc.rs | 8 +- slatedb/src/wal/admin.rs | 210 +++++++++ slatedb/src/wal/gc.rs | 4 +- slatedb/src/wal/mod.rs | 64 ++- 7 files changed, 775 insertions(+), 158 deletions(-) create mode 100644 slatedb/src/wal/admin.rs diff --git a/rfcs/0030-pluggable-wal.md b/rfcs/0030-pluggable-wal.md index 4d5029db4c..3886aae20a 100644 --- a/rfcs/0030-pluggable-wal.md +++ b/rfcs/0030-pluggable-wal.md @@ -393,7 +393,7 @@ pub trait WalReader { /// API for plugging into WAL GC #[async_trait] -pub trait WalGC { +pub trait WalGc { /// Hook for garbage collecting the WAL. Takes a list of ranges of WAL Files that are currently /// referenced by some active Manifest. The implementation may delete any WAL File that is not /// included in the ranges in this list. @@ -402,6 +402,65 @@ pub trait WalGC { referenced_ranges: Vec, ) -> Result<(), WalError>; } + +/// Administrative operations for a WAL implementation. +#[async_trait] +pub trait WalAdmin: Send + Sync + 'static { + /// Creates a garbage collector scoped to the WAL at `path`. + /// + /// ## Arguments + /// - `path`: The database path whose WAL should be garbage collected. + /// + /// ## Returns + /// A garbage collector that can remove unreferenced WAL files at `path`. + fn garbage_collector(&self, path: &Path) -> Box; + + /// Deletes the WAL at `path`. + /// + /// ## Arguments + /// - `path`: The database path whose WAL should be deleted. + /// + /// ## Returns + /// `Ok(())` after the WAL has been deleted, or a [`WalError`] if deletion fails. + async fn delete_wal(&self, path: &Path) -> Result<(), WalError>; + + /// Given a path and WAL ID range, returns true if the WAL at that path is empty within the + /// specified range. A WAL is empty if it holds no records. + /// + /// ## Arguments + /// - `path`: The database path containing the WAL. + /// - `replay_after_wal_id`: The exclusive lower bound of the WAL range to inspect. + /// - `wal_id_last_seen`: The inclusive upper bound of the WAL range to inspect. + /// + /// ## Returns + /// `Ok(true)` if the referenced WAL contains no records, `Ok(false)` if it contains records, + /// or a [`WalError`] if the WAL could not be inspected. + async fn is_empty( + &self, + path: &Path, + replay_after_wal_id: u64, + wal_id_last_seen: u64, + ) -> Result; + + /// Given a source path and manifest, copy the referenced WAL to a destination path and return + /// a replay range. This call must be idempotent (TODO: clarify) + /// + /// ## Arguments + /// - `from_path`: The db path that holds the source WAL range to be copied + /// - `from_manifest`: The source manifest that identifies the WAL to copy + /// - `to_path`: The db path of the clone that the WAL is being copied to. + /// + /// ## Returns + /// A (u64, u64) pair. The first item will be used as the replay start point (exclusive). The + /// second item should be the id of the last WAL file id in the copied WAL. + async fn clone_wal( + &self, + from_path: &Path, + from_manifest: VersionedManifest, + to_path: &Path, + ) -> Result<(u64, u64), WalError>; +} + ``` Users can configure a custom WAL for the writer and reader using the db Builder: @@ -618,6 +677,14 @@ impl WalReader for ObjectStoreWalReader { } ``` +#### Clones + +Clone creation delegates to `WalAdmin::empty` to introspect source WALs to determine if they are +empty when validating that unions/projections don't require copying the WAL. + +Clone creation delegates to `WalAdmin::clone_wal` to copy the WAL from the source db to the clone +db. + #### Error Handling Custom WAL implementations are expected to manage the lifecycle of any background tasks and diff --git a/slatedb/src/clone.rs b/slatedb/src/clone.rs index d9d65041a5..6e40e0bb30 100644 --- a/slatedb/src/clone.rs +++ b/slatedb/src/clone.rs @@ -3,19 +3,16 @@ use crate::checkpoint::Checkpoint; use crate::config::CheckpointOptions; use crate::db::builder::CloneSourceSpec; -use crate::db_state::SsTableId; use crate::error::SlateDBError; use crate::error::SlateDBError::CheckpointMissing; use crate::manifest::store::{ManifestStore, StoredManifest}; -use crate::manifest::{Manifest, ManifestCore, ProjectionConfig}; -use crate::object_stores::ObjectStoreType::{Main, Wal}; -use crate::object_stores::ObjectStores; -use crate::paths::PathResolver; +use crate::manifest::{Manifest, ProjectionConfig, VersionedManifest}; use crate::utils::IdGenerator; +use crate::wal::WalAdmin; use bytes::Bytes; use fail_parallel::{fail_point, FailPointRegistry}; use object_store::path::Path; -use object_store::{ObjectStore, ObjectStoreExt}; +use object_store::ObjectStore; use slatedb_common::clock::SystemClock; use slatedb_common::DbRand; use std::ops::RangeBounds; @@ -35,10 +32,22 @@ pub(crate) type SegmentFilterFn = Arc bool + Send + Sync>; pub(crate) type SegmentProjectionFn = Arc Result + Send + Sync>; +struct CopyWalParams { + from_path: Path, + from_manifest: VersionedManifest, + to_path: Path, +} + +struct CreateCloneManifestResult { + clone_manifest: StoredManifest, + copy_wal_params: Option, +} + pub(crate) async fn create_clone, R: RangeBounds + Clone>( clone_sources: Vec>, clone_path: P, - object_stores: ObjectStores, + object_store: Arc, + wal_admin: Arc, fp_registry: Arc, system_clock: Arc, rand: Arc, @@ -48,40 +57,37 @@ pub(crate) async fn create_clone, R: RangeBounds + Clone>( ) -> Result<(), SlateDBError> { let clone_path = clone_path.into(); - validate_clone_source_specs(clone_sources.clone(), clone_path.clone())?; + validate_clone_source_specs(&clone_sources, &clone_path)?; - let mut clone_manifest = create_clone_manifest( + let CreateCloneManifestResult { + mut clone_manifest, + copy_wal_params, + } = create_clone_manifest( clone_path.clone(), - clone_sources.clone(), - object_stores.store_of(Main).clone(), - object_stores.store_of(Wal).clone(), + clone_sources, + object_store, system_clock.clone(), rand, fp_registry.clone(), projection_range, segment_filter, segment_projection, + wal_admin.as_ref(), ) .await?; if !clone_manifest.db_state().initialized { - // Copy WAL SSTs from all sources - WAL is only supported for single source - // this invariant is enforced in create_clone_manifest() - if clone_sources.len() == 1 { - for source in &clone_sources { - let parent_path = source.path.clone(); - copy_wal_ssts( - object_stores.store_of(Wal).clone(), - clone_manifest.db_state(), - &parent_path, - &clone_path, - fp_registry.clone(), - ) - .await?; - } - } + let (replay_after_wal_id, wal_id_last_seen) = match copy_wal_params { + Some(params) => copy_wal(wal_admin.as_ref(), params).await?, + None => (0, 0), + }; + let next_wal_sst_id = wal_id_last_seen + .checked_add(1) + .ok_or(SlateDBError::InvalidDBState)?; let mut dirty = clone_manifest.prepare_dirty()?; + dirty.value.core.replay_after_wal_id = replay_after_wal_id; + dirty.value.core.next_wal_sst_id = next_wal_sst_id; dirty.value.core.initialized = true; clone_manifest.update(dirty).await?; } @@ -93,17 +99,17 @@ async fn create_clone_manifest + Clone>( clone_path: Path, source_specs: Vec>, object_store: Arc, - wal_object_store: Arc, system_clock: Arc, rand: Arc, #[allow(unused)] fp_registry: Arc, projection_range: Option, segment_filter: Option, segment_projection: Option, -) -> Result { + wal_admin: &dyn WalAdmin, +) -> Result { let clone_manifest_store = Arc::new(ManifestStore::new(&clone_path, object_store.clone())); - let clone_manifest = + let (clone_manifest, copy_wal_params) = match StoredManifest::try_load(clone_manifest_store.clone(), system_clock.clone()).await? { Some(initialized_clone_manifest) if initialized_clone_manifest.db_state().initialized => @@ -122,7 +128,10 @@ async fn create_clone_manifest + Clone>( ) .await?; } - return Ok(initialized_clone_manifest); + return Ok(CreateCloneManifestResult { + clone_manifest: initialized_clone_manifest, + copy_wal_params: None, + }); } Some(uninitialized_clone_manifest) => { for source_spec in &source_specs { @@ -132,7 +141,24 @@ async fn create_clone_manifest + Clone>( &uninitialized_clone_manifest, )?; } - uninitialized_clone_manifest + let copy_wal_params = match &source_specs[..] { + [source_spec] => { + let source = rebuild_source( + source_spec, + &uninitialized_clone_manifest, + &object_store, + &system_clock, + &rand, + &projection_range, + segment_filter.as_ref(), + segment_projection.as_ref(), + ) + .await?; + Some(copy_wal_params_for_source(&source, &clone_path)) + } + _ => None, + }; + (uninitialized_clone_manifest, copy_wal_params) } None => { let sources = build_sources( @@ -145,40 +171,48 @@ async fn create_clone_manifest + Clone>( segment_projection.as_ref(), ) .await?; + let copy_wal_params = match &sources[..] { + [source] => Some(copy_wal_params_for_source(source, &clone_path)), + _ => None, + }; let projection_requested = projection_range.is_some() || segment_filter.is_some() || segment_projection.is_some() || source_specs.iter().any(|s| s.projection_range.is_some()); - let manifest: Manifest = match &sources[..] { + let mut manifest: Manifest = match &sources[..] { [single_source] => { // WAL SSTs are copied to the clone verbatim and replayed in full // when the clone is opened, so entries outside the projected // range would leak into the clone. So we reject projections if // there are non-fence WALs to copy. if projection_requested { - validate_no_data_wal(&sources, &wal_object_store).await?; + validate_no_data_wal(&sources, wal_admin).await?; } Manifest::cloned( &single_source.manifest, single_source.path.to_string(), single_source.checkpoint.id, - rand, + rand.clone(), ) } [..] => { - validate_no_data_wal(&sources, &wal_object_store).await?; - Manifest::cloned_from_union(sources, rand)? + validate_no_data_wal(&sources, wal_admin).await?; + Manifest::cloned_from_union(sources, rand.clone())? } }; + manifest.core.initialized = false; - StoredManifest::store_uninitialized_clone( - clone_manifest_store, - manifest, - system_clock.clone(), + ( + StoredManifest::store_uninitialized_clone( + clone_manifest_store, + manifest, + system_clock.clone(), + ) + .await?, + copy_wal_params, ) - .await? } }; @@ -224,7 +258,10 @@ async fn create_clone_manifest + Clone>( } } - Ok(clone_manifest) + Ok(CreateCloneManifestResult { + clone_manifest, + copy_wal_params, + }) } fn to_byte_range + Clone>(bounds: &T) -> BytesRange { @@ -238,6 +275,12 @@ pub(crate) struct CloneSource { pub checkpoint: Checkpoint, } +impl CloneSource { + fn versioned_manifest(&self) -> VersionedManifest { + VersionedManifest::from_manifest(self.checkpoint.manifest_id, self.manifest.clone()) + } +} + /// Builds a list of clone sources from the provided specifications. For each source spec, a /// manifest at the specified checkpoint is loaded (if the checkpoint is not specified then it is /// created). Additionally, if any of `projection_range`, `segment_filter`, or @@ -254,40 +297,115 @@ async fn build_sources + Clone>( ) -> Result, SlateDBError> { let mut result: Vec = vec![]; for source in source_specs { - let manifest_store = Arc::new(ManifestStore::new(&source.path, object_store.clone())); - let mut latest_manifest = - load_initialized_manifest(manifest_store.clone(), system_clock.clone()).await?; - let checkpoint = - get_or_create_parent_checkpoint(&mut latest_manifest, source.checkpoint, rand.clone()) - .await?; - let mut manifest_at_checkpoint = - manifest_store.read_manifest(checkpoint.manifest_id).await?; - - let range: Option = match (source.projection_range.clone(), projection_range) { - (Some(l), Some(r)) => to_byte_range(&l).intersect(&to_byte_range(r)), - (Some(l), None) => Some(to_byte_range(&l)), - (None, Some(r)) => Some(to_byte_range(r)), - (None, None) => None, - }; + result.push( + build_source( + source, + source.checkpoint, + object_store, + system_clock, + rand, + projection_range, + segment_filter, + segment_projection, + ) + .await?, + ); + } + Ok(result) +} - let config = ProjectionConfig { - global_range: range, - segment_filter: segment_filter.cloned(), - segment_projection: segment_projection.cloned(), - }; - manifest_at_checkpoint = if config.is_noop() { - manifest_at_checkpoint - } else { - Manifest::projected(&manifest_at_checkpoint, &config)? - }; +async fn build_source + Clone>( + source: &CloneSourceSpec, + checkpoint_id: Option, + object_store: &Arc, + system_clock: &Arc, + rand: &Arc, + projection_range: &Option, + segment_filter: Option<&SegmentFilterFn>, + segment_projection: Option<&SegmentProjectionFn>, +) -> Result { + let manifest_store = Arc::new(ManifestStore::new(&source.path, object_store.clone())); + let mut latest_manifest = + load_initialized_manifest(manifest_store.clone(), system_clock.clone()).await?; + let checkpoint = + get_or_create_parent_checkpoint(&mut latest_manifest, checkpoint_id, rand.clone()).await?; + let mut manifest_at_checkpoint = manifest_store.read_manifest(checkpoint.manifest_id).await?; + + let range: Option = match (source.projection_range.clone(), projection_range) { + (Some(l), Some(r)) => to_byte_range(&l).intersect(&to_byte_range(r)), + (Some(l), None) => Some(to_byte_range(&l)), + (None, Some(r)) => Some(to_byte_range(r)), + (None, None) => None, + }; - result.push(CloneSource { - path: source.path.clone(), - manifest: manifest_at_checkpoint, - checkpoint, - }); + let config = ProjectionConfig { + global_range: range, + segment_filter: segment_filter.cloned(), + segment_projection: segment_projection.cloned(), + }; + manifest_at_checkpoint = if config.is_noop() { + manifest_at_checkpoint + } else { + Manifest::projected(&manifest_at_checkpoint, &config)? + }; + + Ok(CloneSource { + path: source.path.clone(), + manifest: manifest_at_checkpoint, + checkpoint, + }) +} + +fn copy_wal_params_for_source(source: &CloneSource, to_path: &Path) -> CopyWalParams { + CopyWalParams { + from_path: source.path.clone(), + from_manifest: source.versioned_manifest(), + to_path: to_path.clone(), } - Ok(result) +} + +async fn rebuild_source + Clone>( + source_spec: &CloneSourceSpec, + clone_manifest: &StoredManifest, + object_store: &Arc, + system_clock: &Arc, + rand: &Arc, + projection_range: &Option, + segment_filter: Option<&SegmentFilterFn>, + segment_projection: Option<&SegmentProjectionFn>, +) -> Result { + // `Manifest::cloned` appends the direct parent after inherited external DBs. Search in reverse + // so a parent that also appears in its own ancestry still resolves to the direct source. + let source_path = source_spec.path.to_string(); + let external_db = clone_manifest + .manifest() + .external_dbs + .iter() + .rev() + .find(|external_db| external_db.path == source_path) + .ok_or(SlateDBError::CloneExternalDbMissing)?; + let manifest_store = Arc::new(ManifestStore::new(&source_spec.path, object_store.clone())); + let latest_manifest = load_initialized_manifest(manifest_store, system_clock.clone()).await?; + let checkpoint_id = external_db + .final_checkpoint_id + .filter(|checkpoint_id| { + latest_manifest + .db_state() + .find_checkpoint(*checkpoint_id) + .is_some() + }) + .unwrap_or(external_db.source_checkpoint_id); + build_source( + source_spec, + Some(checkpoint_id), + object_store, + system_clock, + rand, + projection_range, + segment_filter, + segment_projection, + ) + .await } // Get a checkpoint and the corresponding manifest that will be used as the source @@ -324,16 +442,16 @@ async fn get_or_create_parent_checkpoint( } fn validate_clone_source_specs + Clone>( - specs: Vec>, - clone_path: Path, + specs: &[CloneSourceSpec], + clone_path: &Path, ) -> Result<(), SlateDBError> { if specs.is_empty() { return Err(SlateDBError::InvalidUnionSetEmpty()); } let mut seen_paths = std::collections::HashSet::new(); - for source in &specs { - if clone_path == source.path { + for source in specs { + if clone_path == &source.path { return Err(SlateDBError::IdenticalClonePaths(clone_path.clone())); } if !seen_paths.insert(source.path.to_string()) { @@ -345,37 +463,21 @@ fn validate_clone_source_specs + Clone>( async fn validate_no_data_wal( sources: &[CloneSource], - wal_object_store: &Arc, + wal_admin: &dyn WalAdmin, ) -> Result<(), SlateDBError> { let mut parents_with_wal = vec![]; for source in sources { - let core = &source.manifest.core; - // Cheap manifest check first: if the WAL range is empty there is - // nothing to inspect. - if core.next_wal_sst_id.saturating_sub(1) <= core.replay_after_wal_id { - continue; - } - - let path_resolver = PathResolver::from_root(source.path.clone()); - let mut has_data_wal = false; - for wal_id in (core.replay_after_wal_id + 1)..core.next_wal_sst_id { - let path = path_resolver.sst_path(&SsTableId::Wal(wal_id)); - match wal_object_store.head(&path).await { - Ok(meta) => { - // Fence WALs are zero-byte `SsTableId::Wal` objects (written via - // `TableStore::write_wal_fence`). They contain no data, so dropping them loses - // nothing and the source is allowed to participate in the union. We use the - // same zero-size discriminator that fence garbage_collector/wal_gc.rs uses. - if meta.size > 0 { - has_data_wal = true; - break; - } - } - Err(e) => return Err(SlateDBError::from(e)), - } - } - - if has_data_wal { + let replay_after_wal_id = source.manifest.core.replay_after_wal_id; + let wal_id_last_seen = source + .manifest + .core + .next_wal_sst_id + .checked_sub(1) + .ok_or(SlateDBError::InvalidDBState)?; + if !wal_admin + .is_empty(&source.path, replay_after_wal_id, wal_id_last_seen) + .await? + { parents_with_wal.push(source.path.clone()); } } @@ -467,36 +569,24 @@ async fn load_initialized_manifest( Ok(manifest) } -async fn copy_wal_ssts( - object_store: Arc, - parent_checkpoint_state: &ManifestCore, - parent_path: &Path, - clone_path: &Path, - #[allow(unused)] fp_registry: Arc, -) -> Result<(), SlateDBError> { - let parent_path_resolver = PathResolver::from_root(parent_path.clone()); - let clone_path_resolver = PathResolver::from_root(clone_path.clone()); - - let mut wal_id = parent_checkpoint_state.replay_after_wal_id + 1; - while wal_id < parent_checkpoint_state.next_wal_sst_id { - fail_point!(fp_registry.clone(), "copy-wal-ssts-io-error", |_| Err( - SlateDBError::from(std::io::Error::other("oops")) - )); - - let id = SsTableId::Wal(wal_id); - let parent_path = parent_path_resolver.sst_path(&id); - let clone_path = clone_path_resolver.sst_path(&id); - object_store - .as_ref() - .copy(&parent_path, &clone_path) - .await?; - wal_id += 1; - } - Ok(()) +async fn copy_wal( + wal_admin: &dyn WalAdmin, + params: CopyWalParams, +) -> Result<(u64, u64), SlateDBError> { + let CopyWalParams { + from_path, + from_manifest, + to_path, + } = params; + wal_admin + .clone_wal(&from_path, from_manifest, &to_path) + .await + .map_err(Into::into) } #[cfg(test)] mod tests { + use super::{SegmentFilterFn, SegmentProjectionFn}; use crate::config::{ CheckpointOptions, CheckpointScope, FlushOptions, FlushType, PutOptions, Settings, WriteOptions, @@ -509,12 +599,15 @@ mod tests { use crate::iter::IterationOrder; use crate::manifest::store::{ManifestStore, StoredManifest}; use crate::manifest::Manifest; - use crate::manifest::ManifestCore; + use crate::manifest::{ManifestCore, VersionedManifest}; use crate::object_stores::ObjectStores; use crate::paths::PathResolver; use crate::proptest_util::{rng, sample}; use crate::test_utils; use crate::utils::IdGenerator; + use crate::wal::admin::SlateDbWalAdmin; + use crate::wal::{WalAdmin, WalError, WalFileRange, WalGc}; + use async_trait::async_trait; use bytes::Bytes; use fail_parallel::FailPointRegistry; use object_store::memory::InMemory; @@ -531,6 +624,86 @@ mod tests { use std::sync::Arc; use uuid::Uuid; + struct RemappingWalAdmin { + replay_range: (u64, u64), + expected_manifest_id: Option, + } + + struct NoopWalGc; + + #[async_trait] + impl WalGc for NoopWalGc { + async fn collect(&self, _referenced_ranges: Vec) -> Result<(), WalError> { + Ok(()) + } + } + + #[async_trait] + impl WalAdmin for RemappingWalAdmin { + fn garbage_collector(&self, _path: &Path) -> Box { + Box::new(NoopWalGc) + } + + async fn delete_wal(&self, _path: &Path) -> Result<(), WalError> { + Ok(()) + } + + async fn is_empty( + &self, + _path: &Path, + _replay_after_wal_id: u64, + _wal_id_last_seen: u64, + ) -> Result { + Ok(true) + } + + async fn clone_wal( + &self, + _from_path: &Path, + from_manifest: VersionedManifest, + _to_path: &Path, + ) -> Result<(u64, u64), WalError> { + if let Some(expected_manifest_id) = self.expected_manifest_id { + assert_eq!(from_manifest.id(), expected_manifest_id); + } + Ok(self.replay_range) + } + } + + async fn create_native_clone, R: RangeBounds + Clone>( + clone_sources: Vec>, + clone_path: P, + object_stores: ObjectStores, + fp_registry: Arc, + system_clock: Arc, + rand: Arc, + projection_range: Option, + segment_filter: Option, + segment_projection: Option, + ) -> Result<(), SlateDBError> { + let wal_admin = Arc::new(SlateDbWalAdmin::new( + object_stores + .store_of(crate::object_stores::ObjectStoreType::Wal) + .clone(), + fp_registry.clone(), + )); + crate::clone::create_clone( + clone_sources, + clone_path, + object_stores + .store_of(crate::object_stores::ObjectStoreType::Main) + .clone(), + wal_admin, + fp_registry, + system_clock, + rand, + projection_range, + segment_filter, + segment_projection, + ) + .await + } + // helper method for tests that creates CloneSourceSpec async fn create_clone>( clone_path: P, @@ -546,7 +719,7 @@ mod tests { Some(cp) => CloneSourceSpec::with_checkpoint(parent_path, cp), None => CloneSourceSpec::new(parent_path), }; - crate::clone::create_clone( + create_native_clone( vec![source], clone_path, ObjectStores::new(object_store, Some(wal_object_store)), @@ -560,6 +733,105 @@ mod tests { .await } + #[tokio::test] + async fn should_stamp_wal_range_returned_by_wal_admin() { + let object_store: Arc = Arc::new(InMemory::new()); + let parent_path = Path::from("/tmp/test_parent_remapped_wal"); + let clone_path = Path::from("/tmp/test_clone_remapped_wal"); + let system_clock: Arc = Arc::new(DefaultSystemClock::new()); + + let mut parent_manifest = StoredManifest::create_new_db( + Arc::new(ManifestStore::new(&parent_path, object_store.clone())), + ManifestCore::new(), + system_clock.clone(), + ) + .await + .unwrap(); + let checkpoint = parent_manifest + .write_checkpoint(Uuid::new_v4(), &CheckpointOptions::default()) + .await + .unwrap(); + + let wal_admin = RemappingWalAdmin { + replay_range: (41, 46), + expected_manifest_id: Some(checkpoint.manifest_id), + }; + let source: CloneSourceSpec = CloneSourceSpec::with_checkpoint(parent_path, checkpoint.id); + crate::clone::create_clone( + vec![source], + clone_path.clone(), + object_store.clone(), + Arc::new(wal_admin), + Arc::new(FailPointRegistry::new()), + system_clock.clone(), + Arc::new(DbRand::default()), + None, + None, + None, + ) + .await + .unwrap(); + + let manifest = StoredManifest::load( + Arc::new(ManifestStore::new(&clone_path, object_store)), + system_clock, + ) + .await + .unwrap(); + assert!(manifest.db_state().initialized); + assert_eq!(manifest.db_state().replay_after_wal_id, 41); + assert_eq!(manifest.db_state().next_wal_sst_id, 47); + } + + #[tokio::test] + async fn should_reset_wal_range_when_clone_does_not_copy_wal() { + let object_store: Arc = Arc::new(InMemory::new()); + let parent_paths = [ + Path::from("/tmp/test_parent_no_wal_a"), + Path::from("/tmp/test_parent_no_wal_b"), + ]; + let clone_path = Path::from("/tmp/test_clone_no_wal"); + let system_clock: Arc = Arc::new(DefaultSystemClock::new()); + + for parent_path in &parent_paths { + StoredManifest::create_new_db( + Arc::new(ManifestStore::new(parent_path, object_store.clone())), + ManifestCore::new(), + system_clock.clone(), + ) + .await + .unwrap(); + } + + crate::clone::create_clone( + parent_paths.into_iter().map(CloneSourceSpec::new).collect(), + clone_path.clone(), + object_store.clone(), + Arc::new(RemappingWalAdmin { + replay_range: (41, 47), + expected_manifest_id: None, + }), + Arc::new(FailPointRegistry::new()), + system_clock.clone(), + Arc::new(DbRand::default()), + None, + None, + None, + ) + .await + .unwrap(); + + let manifest = StoredManifest::load( + Arc::new(ManifestStore::new(&clone_path, object_store)), + system_clock, + ) + .await + .unwrap(); + assert!(manifest.db_state().initialized); + assert_eq!(manifest.db_state().replay_after_wal_id, 0); + assert_eq!(manifest.db_state().next_wal_sst_id, 1); + } + #[tokio::test] async fn should_clone_latest_state_if_no_checkpoint_provided() { let mut rng = rng::new_test_rng(None); @@ -1097,7 +1369,7 @@ mod tests { ) .await .unwrap_err(); - assert!(matches!(err, SlateDBError::IoError(_))); + assert!(matches!(err, SlateDBError::WalUnavailable(_))); fail_parallel::cfg(Arc::clone(&fp_registry), "copy-wal-ssts-io-error", "off").unwrap(); create_clone( @@ -1358,10 +1630,11 @@ mod tests { .unwrap_err(); assert!(matches!( err, - SlateDBError::ObjectStoreError(ref source) + SlateDBError::WalUnavailable(ref source) if matches!( - source.as_ref(), - ObjectStoreError::NotFound { path, .. } if path == &expected_missing_wal_path + source.downcast_ref::(), + Some(ObjectStoreError::NotFound { path, .. }) + if path == &expected_missing_wal_path ) )); } @@ -1442,7 +1715,7 @@ mod tests { Bound::Included(Bytes::from_static(b"aaa")), Bound::Excluded(Bytes::from_static(b"bbb")), ); - let err = crate::clone::create_clone( + let err = create_native_clone( vec![CloneSourceSpec::new(parent_path.clone())], clone_path.clone(), ObjectStores::new(object_store.clone(), Some(object_store.clone())), @@ -1475,7 +1748,7 @@ mod tests { .unwrap(); parent_db.close().await.unwrap(); - crate::clone::create_clone( + create_native_clone( vec![CloneSourceSpec::new(parent_path.clone())], clone_path.clone(), ObjectStores::new(object_store.clone(), Some(object_store.clone())), @@ -1591,7 +1864,7 @@ mod tests { object_store: Arc, projection: Option, ) { - crate::clone::create_clone( + create_native_clone( sources, clone_path.clone(), ObjectStores::new(object_store.clone(), Some(object_store)), @@ -2235,7 +2508,7 @@ mod tests { // Source B has no extra WAL. build_plain_wal_disabled_parent(&parent_path_b, object_store.clone(), &table_b).await; - crate::clone::create_clone( + create_native_clone( vec![ CloneSourceSpec::new(parent_path_a.clone()), CloneSourceSpec::new(parent_path_b.clone()), @@ -2298,7 +2571,7 @@ mod tests { .await; build_plain_wal_disabled_parent(&parent_path_b, object_store.clone(), &table_b).await; - let err = crate::clone::create_clone( + let err = create_native_clone( vec![ CloneSourceSpec::new(parent_path_a.clone()), CloneSourceSpec::new(parent_path_b.clone()), @@ -2367,7 +2640,7 @@ mod tests { })) .to_string(); - let err = crate::clone::create_clone( + let err = create_native_clone( vec![ CloneSourceSpec::new(parent_path_a.clone()), CloneSourceSpec::new(parent_path_b.clone()), @@ -2387,10 +2660,10 @@ mod tests { assert!( matches!( err, - SlateDBError::ObjectStoreError(ref source) + SlateDBError::WalUnavailable(ref source) if matches!( - source.as_ref(), - ObjectStoreError::NotFound { path, .. } + source.downcast_ref::(), + Some(ObjectStoreError::NotFound { path, .. }) if path == &expected_missing_wal_path ) ), diff --git a/slatedb/src/db/builder.rs b/slatedb/src/db/builder.rs index e427e56aad..ffb103d814 100644 --- a/slatedb/src/db/builder.rs +++ b/slatedb/src/db/builder.rs @@ -161,6 +161,7 @@ use crate::retrying_object_store::RetryingObjectStore; use crate::tablestore::{TableStore, TableStoreKind}; use crate::utils::SafeSender; use crate::utils::WatchableOnceCell; +use crate::wal::admin::SlateDbWalAdmin; use crate::wal::wal_disabled::DisabledWalObserver; use crate::wal::WalObserver; use slatedb_common::clock::DefaultSystemClock; @@ -2054,11 +2055,17 @@ impl + Clone> CloneBuilder { /// Build and execute the clone operation. pub async fn build(self) -> Result<(), crate::Error> { + let fp_registry = Arc::new(FailPointRegistry::new()); + let wal_admin = Arc::new(SlateDbWalAdmin::new( + self.wal_object_store.unwrap_or(self.object_store.clone()), + fp_registry.clone(), + )); crate::clone::create_clone( self.sources, self.clone_path, - ObjectStores::new(self.object_store, self.wal_object_store), - Arc::new(FailPointRegistry::new()), + self.object_store, + wal_admin, + fp_registry, self.system_clock .unwrap_or_else(|| Arc::new(DefaultSystemClock::new())), self.rand.unwrap_or_else(|| Arc::new(Default::default())), diff --git a/slatedb/src/garbage_collector/wal_gc.rs b/slatedb/src/garbage_collector/wal_gc.rs index 2ef3be417d..3b455ed25c 100644 --- a/slatedb/src/garbage_collector/wal_gc.rs +++ b/slatedb/src/garbage_collector/wal_gc.rs @@ -2,7 +2,7 @@ use crate::manifest::Manifest; use crate::{ error::SlateDBError, manifest::store::ManifestStore, - wal::{WalFileRange, WalGC}, + wal::{WalFileRange, WalGc}, }; use chrono::{DateTime, Utc}; use std::collections::BTreeMap; @@ -14,7 +14,7 @@ use super::GcTask; #[derive(Clone)] pub(crate) struct WalGcTask { manifest_store: Arc, - wal_gc: Arc, + wal_gc: Arc, resource: &'static str, } @@ -29,7 +29,7 @@ impl std::fmt::Debug for WalGcTask { impl WalGcTask { pub(super) fn new( manifest_store: Arc, - wal_gc: Arc, + wal_gc: Arc, resource: &'static str, ) -> Self { Self { @@ -114,7 +114,7 @@ mod tests { } #[async_trait] - impl WalGC for RecordingWalGc { + impl WalGc for RecordingWalGc { async fn collect(&self, referenced_ranges: Vec) -> Result<(), WalError> { self.calls.lock().unwrap().push(referenced_ranges); Ok(()) diff --git a/slatedb/src/wal/admin.rs b/slatedb/src/wal/admin.rs new file mode 100644 index 0000000000..7d7546a6cb --- /dev/null +++ b/slatedb/src/wal/admin.rs @@ -0,0 +1,210 @@ +use crate::block_cache_policy::BlockCachePolicy; +use crate::config::GarbageCollectorDirectoryOptions; +use crate::db_state::SsTableId; +use crate::format::sst::SsTableFormat; +use crate::garbage_collector::stats::GcStats; +use crate::object_stores::ObjectStores; +use crate::paths::PathResolver; +use crate::tablestore::{TableStore, TableStoreKind}; +use crate::wal::gc::{SlateDbWalGc, WalGcMode}; +use crate::wal::{WalAdmin, WalError, WalGc}; +use crate::VersionedManifest; +use async_trait::async_trait; +use fail_parallel::{fail_point, FailPointRegistry}; +use futures::StreamExt; +use object_store::path::Path; +use object_store::{ObjectStore, ObjectStoreExt}; +use slatedb_common::clock::DefaultSystemClock; +use slatedb_common::metrics::MetricsRecorderHelper; +use std::sync::Arc; + +#[derive(Clone)] +pub(crate) struct SlateDbWalAdmin { + object_store: Arc, + #[cfg_attr(not(test), allow(dead_code))] + fp_registry: Arc, +} + +impl SlateDbWalAdmin { + pub(crate) fn new( + object_store: Arc, + fp_registry: Arc, + ) -> Self { + Self { + object_store, + fp_registry, + } + } + + fn replay_range(manifest: &VersionedManifest) -> Result<(u64, u64), WalError> { + let replay_after_wal_id = manifest.replay_after_wal_id(); + let wal_id_last_seen = manifest + .next_wal_sst_id() + .checked_sub(1) + .ok_or_else(Self::invalid_manifest)?; + Ok((replay_after_wal_id, wal_id_last_seen)) + } + + fn invalid_manifest() -> WalError { + WalError::InternalError(Arc::new(std::io::Error::new( + std::io::ErrorKind::InvalidInput, + "source manifest must have a positive next WAL file ID", + ))) + } + + fn has_wal_file_ids(replay_after_wal_id: u64, wal_id_last_seen: u64) -> bool { + wal_id_last_seen > replay_after_wal_id + } + + async fn paths_under(&self, path: &Path) -> Result, WalError> { + let mut objects = self.object_store.list(Some(path)); + let mut paths = Vec::new(); + while let Some(object) = objects.next().await { + let object = object.map_err(|err| WalError::Unavailable(Arc::new(err)))?; + paths.push(object.location); + } + Ok(paths) + } +} + +#[async_trait] +impl WalAdmin for SlateDbWalAdmin { + fn garbage_collector(&self, path: &Path) -> Box { + let table_store = Arc::new(TableStore::new( + ObjectStores::new(self.object_store.clone(), None), + SsTableFormat::default(), + path.clone(), + None, + TableStoreKind::GC, + BlockCachePolicy::default(), + )); + Box::new(SlateDbWalGc::new( + table_store, + Arc::new(GcStats::new(&MetricsRecorderHelper::noop())), + GarbageCollectorDirectoryOptions::default(), + WalGcMode::Regular, + None, + Arc::new(DefaultSystemClock::new()), + )) + } + + async fn delete_wal(&self, path: &Path) -> Result<(), WalError> { + // Collect the paths first so listing is complete before objects are removed. + let wal_path = PathResolver::from_root(path.clone()).wal_path(); + for object_path in self.paths_under(&wal_path).await? { + self.object_store + .delete(&object_path) + .await + .map_err(|err| WalError::Unavailable(Arc::new(err)))?; + } + Ok(()) + } + + async fn is_empty( + &self, + path: &Path, + replay_after_wal_id: u64, + wal_id_last_seen: u64, + ) -> Result { + // Avoid object-store requests when the manifest's WAL range contains no file IDs. + if !Self::has_wal_file_ids(replay_after_wal_id, wal_id_last_seen) { + return Ok(true); + } + + let path_resolver = PathResolver::from_root(path.clone()); + for wal_id in (replay_after_wal_id + 1)..=wal_id_last_seen { + let path = path_resolver.sst_path(&SsTableId::Wal(wal_id)); + let metadata = self + .object_store + .head(&path) + .await + .map_err(|err| WalError::Unavailable(Arc::new(err)))?; + + // Native SlateDB WAL fences are zero-byte WAL objects and contain no records. + if metadata.size > 0 { + return Ok(false); + } + } + Ok(true) + } + + async fn clone_wal( + &self, + from_path: &Path, + from_manifest: VersionedManifest, + to_path: &Path, + ) -> Result<(u64, u64), WalError> { + let (replay_after_wal_id, wal_id_last_seen) = Self::replay_range(&from_manifest)?; + let from_path_resolver = PathResolver::from_root(from_path.clone()); + let to_path_resolver = PathResolver::from_root(to_path.clone()); + + if Self::has_wal_file_ids(replay_after_wal_id, wal_id_last_seen) { + for wal_id in (replay_after_wal_id + 1)..=wal_id_last_seen { + fail_point!(self.fp_registry.clone(), "copy-wal-ssts-io-error", |_| Err( + WalError::Unavailable(Arc::new(std::io::Error::other("oops"))) + )); + + let id = SsTableId::Wal(wal_id); + let source = from_path_resolver.sst_path(&id); + let destination = to_path_resolver.sst_path(&id); + self.object_store + .as_ref() + .copy(&source, &destination) + .await + .map_err(|err| WalError::Unavailable(Arc::new(err)))?; + } + } + + Ok((replay_after_wal_id, wal_id_last_seen)) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use bytes::Bytes; + use object_store::memory::InMemory; + + #[tokio::test] + async fn delete_wal_deletes_only_objects_under_the_wal_prefix() { + let object_store: Arc = Arc::new(InMemory::new()); + let wal_admin = + SlateDbWalAdmin::new(object_store.clone(), Arc::new(FailPointRegistry::new())); + let db_path = Path::from("db"); + let wal_object = PathResolver::from_root(db_path.clone()).sst_path(&SsTableId::Wal(1)); + let non_wal_object = db_path + .clone() + .join("manifest") + .join("00000000000000000001"); + let sibling_object = Path::from("other/wal/00000000000000000002.sst"); + object_store + .put(&wal_object, Bytes::from_static(b"wal").into()) + .await + .unwrap(); + object_store + .put(&non_wal_object, Bytes::from_static(b"keep").into()) + .await + .unwrap(); + object_store + .put(&sibling_object, Bytes::from_static(b"keep").into()) + .await + .unwrap(); + + wal_admin.delete_wal(&db_path).await.unwrap(); + + assert!(matches!( + object_store.head(&wal_object).await, + Err(object_store::Error::NotFound { .. }) + )); + assert!(object_store.head(&non_wal_object).await.is_ok()); + assert!(object_store.head(&sibling_object).await.is_ok()); + } + + #[test] + fn creates_a_path_scoped_garbage_collector() { + let object_store: Arc = Arc::new(InMemory::new()); + let wal_admin = SlateDbWalAdmin::new(object_store, Arc::new(FailPointRegistry::new())); + + let _collector = wal_admin.garbage_collector(&Path::from("db")); + } +} diff --git a/slatedb/src/wal/gc.rs b/slatedb/src/wal/gc.rs index e583b11b2c..55e4843ce4 100644 --- a/slatedb/src/wal/gc.rs +++ b/slatedb/src/wal/gc.rs @@ -3,7 +3,7 @@ use crate::db_state::SsTableId; use crate::garbage_collector::stats::GcStats; use crate::garbage_collector::{retain_allowed_by_gc_filter, GcFilter, GC_DELETE_CONCURRENCY}; use crate::tablestore::TableStore; -use crate::wal::{WalError, WalFileRange, WalGC}; +use crate::wal::{WalError, WalFileRange, WalGc}; use async_trait::async_trait; use chrono::{DateTime, Utc}; use futures::StreamExt; @@ -155,7 +155,7 @@ impl SlateDbWalGc { } #[async_trait] -impl WalGC for SlateDbWalGc { +impl WalGc for SlateDbWalGc { async fn collect(&self, referenced_ranges: Vec) -> Result<(), WalError> { let utc_now = self.system_clock.now(); let min_age = self.wal_sst_min_age(); diff --git a/slatedb/src/wal/mod.rs b/slatedb/src/wal/mod.rs index 8efa80e570..7b3828b935 100644 --- a/slatedb/src/wal/mod.rs +++ b/slatedb/src/wal/mod.rs @@ -3,11 +3,13 @@ use crate::manifest::store::FenceableManifest; use crate::{CloseReason, ErrorKind, RowEntry, VersionedManifest}; use async_trait::async_trait; use futures::future::BoxFuture; +use object_store::path::Path; use std::error::Error; use std::fmt::{Display, Formatter}; use std::ops::{Bound, Range}; use std::sync::Arc; +pub(crate) mod admin; pub(crate) mod gc; #[cfg(test)] pub(crate) mod test_utils; @@ -282,16 +284,74 @@ pub trait WalReader { /// Trait that defines the contract between SlateDB's garbage collector and a custom WAL /// implementation. SlateDB tracks the set of currently referenced WAL ranges in its manifest. -/// When the Garbage Collector runs, it computes this set and calls [`WalGC::collect`] so that +/// When the Garbage Collector runs, it computes this set and calls [`WalGc::collect`] so that /// the implementation can clean up any un-referenced WAL storage. #[async_trait] -pub trait WalGC: Send + Sync + 'static { +pub trait WalGc: Send + Sync + 'static { /// Hook for garbage collecting the WAL. Takes a list of ranges of WAL Files that are currently /// referenced by some active Manifest. The implementation may delete any WAL File that is not /// included in the ranges in this list. async fn collect(&self, referenced_ranges: Vec) -> Result<(), WalError>; } +/// Administrative operations for a WAL implementation. +#[async_trait] +pub trait WalAdmin: Send + Sync + 'static { + /// Creates a garbage collector scoped to the WAL at `path`. + /// + /// ## Arguments + /// - `path`: The database path whose WAL should be garbage collected. + /// + /// ## Returns + /// A garbage collector that can remove unreferenced WAL files at `path`. + fn garbage_collector(&self, path: &Path) -> Box; + + /// Deletes the WAL at `path`. + /// + /// ## Arguments + /// - `path`: The database path whose WAL should be deleted. + /// + /// ## Returns + /// `Ok(())` after the WAL has been deleted, or a [`WalError`] if deletion fails. + async fn delete_wal(&self, path: &Path) -> Result<(), WalError>; + + /// Given a path and WAL ID range, returns true if the WAL at that path is empty within the + /// specified range. A WAL is empty if it holds no records. + /// + /// ## Arguments + /// - `path`: The database path containing the WAL. + /// - `replay_after_wal_id`: The exclusive lower bound of the WAL range to inspect. + /// - `wal_id_last_seen`: The inclusive upper bound of the WAL range to inspect. + /// + /// ## Returns + /// `Ok(true)` if the referenced WAL contains no records, `Ok(false)` if it contains records, + /// or a [`WalError`] if the WAL could not be inspected. + async fn is_empty( + &self, + path: &Path, + replay_after_wal_id: u64, + wal_id_last_seen: u64, + ) -> Result; + + /// Given a source path and manifest, copy the referenced WAL to a destination path and return + /// a replay range. This call must be idempotent (TODO: clarify) + /// + /// ## Arguments + /// - `from_path`: The db path that holds the source WAL range to be copied + /// - `from_manifest`: The source manifest that identifies the WAL to copy + /// - `to_path`: The db path of the clone that the WAL is being copied to. + /// + /// ## Returns + /// A (u64, u64) pair. The first item will be used as the replay start point (exclusive). The + /// second item should be the id of the last WAL file id in the copied WAL. + async fn clone_wal( + &self, + from_path: &Path, + from_manifest: VersionedManifest, + to_path: &Path, + ) -> Result<(u64, u64), WalError>; +} + impl From for WalError { fn from(status: WalStatus) -> Self { status From bc34de681cf7082de99810e01a002f82822b936a Mon Sep 17 00:00:00 2001 From: Rui Fan <1996fanrui@gmail.com> Date: Tue, 11 Aug 2026 18:07:09 +0200 Subject: [PATCH 28/65] [2004] Do not invoke user metrics callbacks under the db.state write lock (#2006) --- slatedb/src/db.rs | 127 +++++++++++++++++- .../src/memtable_flusher/manifest_writer.rs | 30 ++--- 2 files changed, 138 insertions(+), 19 deletions(-) diff --git a/slatedb/src/db.rs b/slatedb/src/db.rs index 17d75dad2d..7b32750955 100644 --- a/slatedb/src/db.rs +++ b/slatedb/src/db.rs @@ -2263,7 +2263,8 @@ mod tests { use slatedb_common::clock::DefaultSystemClock; use slatedb_common::clock::MockSystemClock; use slatedb_common::metrics::{ - lookup_metric, lookup_metric_with_labels, DefaultMetricsRecorder, MetricValue, + lookup_metric, lookup_metric_with_labels, CounterFn, DefaultMetricsRecorder, GaugeFn, + HistogramFn, MetricValue, MetricsRecorder, NoopMetricsRecorder, UpDownCounterFn, }; use std::collections::BTreeMap; use std::collections::Bound::Included; @@ -10180,6 +10181,130 @@ mod tests { db.close().await.unwrap(); } + struct GaugeBlockControl { + target: &'static str, + armed: AtomicBool, + tripped: AtomicBool, + release: AtomicBool, + entered_tx: tokio::sync::mpsc::UnboundedSender<()>, + } + + struct BlockableGauge { + name: String, + control: Arc, + } + + impl GaugeFn for BlockableGauge { + fn set(&self, _value: i64) { + if self.name != self.control.target { + return; + } + if !self.control.armed.load(Ordering::SeqCst) { + return; + } + if self.control.tripped.swap(true, Ordering::SeqCst) { + return; + } + let _ = self.control.entered_tx.send(()); + while !self.control.release.load(Ordering::SeqCst) { + std::thread::sleep(Duration::from_millis(5)); + } + } + } + + struct BlockingGaugeRecorder { + control: Arc, + } + + impl MetricsRecorder for BlockingGaugeRecorder { + fn register_counter( + &self, + name: &str, + description: &str, + labels: &[(&str, &str)], + ) -> Arc { + NoopMetricsRecorder.register_counter(name, description, labels) + } + + fn register_gauge( + &self, + name: &str, + _description: &str, + _labels: &[(&str, &str)], + ) -> Arc { + Arc::new(BlockableGauge { + name: name.to_string(), + control: self.control.clone(), + }) + } + + fn register_up_down_counter( + &self, + name: &str, + description: &str, + labels: &[(&str, &str)], + ) -> Arc { + NoopMetricsRecorder.register_up_down_counter(name, description, labels) + } + + fn register_histogram( + &self, + name: &str, + description: &str, + labels: &[(&str, &str)], + boundaries: &[f64], + ) -> Arc { + NoopMetricsRecorder.register_histogram(name, description, labels, boundaries) + } + } + + #[tokio::test(flavor = "multi_thread", worker_threads = 4)] + async fn test_should_not_hold_state_lock_during_metrics_callbacks() { + // User metrics callbacks must not run under the `db.state` write + // lock. Block a gauge callback and assert `get()` still completes. + let object_store: Arc = Arc::new(InMemory::new()); + let (entered_tx, mut entered_rx) = tokio::sync::mpsc::unbounded_channel(); + let control = Arc::new(GaugeBlockControl { + target: crate::db_stats::L0_SST_COUNT, + armed: AtomicBool::new(false), + tripped: AtomicBool::new(false), + release: AtomicBool::new(false), + entered_tx, + }); + let recorder = Arc::new(BlockingGaugeRecorder { + control: control.clone(), + }); + let db = Db::builder( + "/tmp/test_should_not_hold_state_lock_during_metrics_callbacks", + object_store, + ) + .with_settings(test_db_options(0, 1024, None)) + .with_metrics_recorder(recorder) + .build() + .await + .unwrap(); + + control.armed.store(true, Ordering::SeqCst); + tokio::time::timeout(Duration::from_secs(10), entered_rx.recv()) + .await + .expect("manifest stats gauge was never set") + .unwrap(); + + let read_task = tokio::spawn({ + let db = db.clone(); + async move { db.get(b"key").await } + }); + let read_result = tokio::time::timeout(Duration::from_secs(5), read_task).await; + control.release.store(true, Ordering::SeqCst); + let value = read_result + .expect("db.get() deadlocked while a metrics callback was in flight") + .unwrap() + .unwrap(); + assert_eq!(value, None); + + db.close().await.unwrap(); + } + #[tokio::test] async fn test_should_record_segment_max_l0_sst_count_with_extractor() { // With a segment extractor configured, `l0_sst_count` sums L0 SSTs diff --git a/slatedb/src/memtable_flusher/manifest_writer.rs b/slatedb/src/memtable_flusher/manifest_writer.rs index 7e1c94557a..babc6bbdd2 100644 --- a/slatedb/src/memtable_flusher/manifest_writer.rs +++ b/slatedb/src/memtable_flusher/manifest_writer.rs @@ -19,10 +19,11 @@ use super::uploader::UploadedMemtable; use crate::checkpoint::CheckpointCreateResult; use crate::config::CheckpointOptions; use crate::db::DbInner; -use crate::db_state::{collect_touched_segments, COWDbState, DbState, SsTableId, SsTableView}; +use crate::db_state::{collect_touched_segments, DbState, SsTableId, SsTableView}; use crate::dispatcher::MessageHandler; use crate::error::SlateDBError; use crate::manifest::store::FenceableManifest; +use crate::manifest::Manifest; use crate::oracle::Oracle; use crate::utils::IdGenerator; use crate::utils::SafeSender; @@ -32,6 +33,7 @@ use bytes::Bytes; use futures::stream::BoxStream; use futures::StreamExt; use parking_lot::RwLockWriteGuard; +use slatedb_txn_obj::DirtyObject; use std::cmp; use std::collections::{BTreeMap, HashSet}; use std::sync::Arc; @@ -620,9 +622,7 @@ impl ManifestWriterHandler { Ok(()) } - fn clone_local_manifest_for_write( - &self, - ) -> slatedb_txn_obj::DirtyObject { + fn clone_local_manifest_for_write(&self) -> DirtyObject { let dirty = { let rguard_state = self.db.state.read(); rguard_state.state().manifest.clone() @@ -655,29 +655,23 @@ impl ManifestWriterHandler { result } - fn merge_remote_manifest( - &self, - remote_dirty: slatedb_txn_obj::DirtyObject, - ) { - let dirty_manifest = { + fn merge_remote_manifest(&self, remote_dirty: DirtyObject) { + let manifest = { let mut wguard_state = self.db.state.write(); wguard_state.merge_remote_manifest(remote_dirty); - let cow = wguard_state.state(); - self.update_stats_for_manifest(&cow); - cow.manifest.clone() + wguard_state.state().manifest.clone() }; - self.db - .status_manager - .report_manifest(dirty_manifest.into()); + self.update_stats_for_manifest(&manifest); + self.db.status_manager.report_manifest(manifest.into()); } - fn update_stats_for_manifest(&self, cow: &COWDbState) { + fn update_stats_for_manifest(&self, manifest: &DirtyObject) { let mut l0_ssts = 0usize; let mut segment_max_l0_ssts = 0usize; let mut sorted_runs = 0usize; let mut sst_views = 0usize; let mut distinct_ssts: HashSet = HashSet::new(); - for tree in cow.core().trees() { + for tree in manifest.value.core.trees() { l0_ssts += tree.l0.len(); // Track the largest single tree: backpressure is driven by `segment_max_l0_sst_count` // because `l0_max_ssts` is enforced per-tree. @@ -705,7 +699,7 @@ impl ManifestWriterHandler { self.db .db_stats .external_db_count - .set(cow.manifest.value.external_dbs.len() as i64); + .set(manifest.value.external_dbs.len() as i64); } async fn write_checkpoint_safely( From 8aa2c10eb9ff3db3519f75f75d10e5487df29805 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Finn=20V=C3=B6lkel?= Date: Tue, 11 Aug 2026 18:24:01 +0200 Subject: [PATCH 29/65] Expose SstFile::index as an opaque view instead of copying data. (#2019) --- rfcs/0020-range-metadata.md | 20 +++-- slatedb/src/lib.rs | 2 +- slatedb/src/partitioned_keyspace.rs | 2 +- slatedb/src/sst_reader.rs | 130 +++++++++++++++++++++++----- 4 files changed, 123 insertions(+), 31 deletions(-) diff --git a/rfcs/0020-range-metadata.md b/rfcs/0020-range-metadata.md index 4563f423a0..6b1d266047 100644 --- a/rfcs/0020-range-metadata.md +++ b/rfcs/0020-range-metadata.md @@ -153,11 +153,16 @@ impl SstFile { /// SSTs that were written before the stats block was added. pub async fn stats(&self) -> Result, crate::Error>; - /// Returns `(block_offset, first_key)` pairs from the SST index block. - pub async fn index(&self) -> Result, crate::Error>; + /// Returns a zero-copy view of the SST index block. + pub async fn index(&self) -> Result; } ``` +`SstIndex` retains the cached index data and provides indexed access, +iteration, and binary-search partition points over borrowed first keys. This +avoids materializing and copying the full index for consumers that only need +to locate blocks. + ```rust pub struct SstStats { pub num_puts: u64, @@ -181,7 +186,7 @@ The `SstFile::info()` call is primarily for users that don't have access to a `M The downside is that `open()` requires a read to obtain the `SsTableHandle` even if the caller only wants to call `metadata()`, which doesn't need it. This is a fine tradeoff. -`index()` calls `SsTableFormat::read_index()`, which reads `info.index_offset..info.index_offset + info.index_len`, decompresses, and returns an `SsTableIndexOwned`. The method materializes `Vec<(u64, Bytes)>` from the FlatBuffer `BlockMeta` entries (each has `offset()` and `first_key()`). Caching uses `DbCache::get_index` / `insert` keyed by `(sst_id, index_offset)`, matching the existing pattern in `TableStore::read_index()`. +`index()` calls `SsTableFormat::read_index()`, which reads `info.index_offset..info.index_offset + info.index_len`, decompresses, and returns an `SsTableIndexOwned`. The returned `SstIndex` retains the cached `Arc` and reads FlatBuffer `BlockMeta` entries without copying their keys. Caching is handled by `TableStore::read_index()`. The existing `SstFileMetadata` struct in `tablestore.rs` (currently `pub(crate)`) is made `pub`. @@ -253,7 +258,7 @@ For cardinality: open each covering SST with `SstReader` (via `view.sst`) and ca #### Refined estimate — block-level for boundary SSTs -Most `SsTableView`s returned by `tables_covering_range()` are fully contained within the query range — their stats apply directly. Only the first and last view in each sorted run partially overlap. For these two boundary SSTs, call `sst_file.index()` to get the index `[(offset, first_key), ...]`. Binary search for the range start key in the first boundary SST to find where the range begins; binary search for the range end key in the last boundary SST to find where it ends. Note that an `SsTableView` may have a `visible_range()` projection that further restricts the effective key range — the query range should be intersected with the view's visible range before performing the binary search. +Most `SsTableView`s returned by `tables_covering_range()` are fully contained within the query range — their stats apply directly. Only the first and last view in each sorted run partially overlap. For these two boundary SSTs, call `sst_file.index()` to get an `SstIndex`. Use `partition_point()` to find the range start in the first boundary SST and the range end in the last boundary SST. Note that an `SsTableView` may have a `visible_range()` projection that further restricts the effective key range — the query range should be intersected with the view's visible range before searching the index. These offsets are compressed/stored sizes since the block index tracks on-disk offsets. @@ -398,7 +403,7 @@ SST stats block: - Record counting requires both stats (for `block_stats`) and index (for binary search on keys). Approximate count: stats + index reads. Exact count: + at most 2 data block reads per boundary SST. `SstFile::index()`: -- One index block read per SST, cacheable via the block cache. No changes to `BlockMeta` format. +- One index block read per SST, cacheable via the block cache. The returned `SstIndex` retains the cached data and exposes borrowed first keys without copying them. No changes to `BlockMeta` format. Memtable metrics via `Db::metrics()`: - No I/O. Reads atomic counters. @@ -435,7 +440,7 @@ Unit tests: - `SstReader::open()`: loading SST footer and constructing `SstFile` - `SstReader::open_with_handle()`: constructing `SstFile` from an existing `SsTableHandle` - `SstFile::stats()`: correct reading and population of `SstStats` from the stats block -- `SstFile::index()`: returns correct `(offset, first_key)` pairs matching the SST's block index +- `SstFile::index()`: returns an `SstIndex` whose accessors expose the correct `(offset, first_key)` pairs and partition points - `block_stats` vector: parallel to index, builder correctly tracks per-block put/delete/merge counts - Backward compatibility: old SSTs without stats return `None` - `Db::manifest()`: returns current manifest state with L0 and sorted runs @@ -502,4 +507,5 @@ Another alternative not explored is sample-based estimation: sample N random blo - **2026-02-16**: Stats fields moved into `SsTableInfo` (in `sst.fbs`) instead of a separate footer block. Removed `SstStats` struct — `SstFile::info()` returns `SsTableInfo` directly. `SstFile` now holds `SsTableHandle` + `Arc`. Added `object_store_cache_options` parameter to `SstReader::new()`. (PR #1220 review feedback from @criccomini). - **2026-02-19**: Reverted to separate stats block approach. Stats fields moved back out of `SsTableInfo` into a dedicated stats block within the SST file, referenced by `stats_offset`/`stats_len` in `SsTableInfo`. Reintroduced `SstStats` struct and `SstFile::stats()` method. This keeps `SsTableInfo` (and the manifest) lean — 16 bytes per SST vs 40 bytes — which matters for large DBs. Added `SstReader::open_with_handle()` for zero-I/O construction from an existing `SsTableHandle`. (PR #1220 review feedback from @rodesai and @criccomini). - **2026-02-25**: Added per-block record counts as `block_stats: [BlockStats]` in `SstStats` (stats block). `BlockStats` contains `num_puts`/`num_deletes`/`num_merges`, mirroring the SST-level aggregate fields. `BlockStats` uses a FlatBuffers `table` for future extensibility. -- **2026-03-19**: Updated RFC to reflect `SsTableView` indirection introduced in [#1362](https://github.com/slatedb/slatedb/pull/1362). `ManifestCore` and `SortedRun` now contain `SsTableView` references instead of raw `SsTableHandle`s. Updated `tables_covering_range()` return type to `VecDeque<&SsTableView>`, replaced `SsTableHandle::estimate_size()`/`visible_range()` references with `SsTableView` equivalents, and added "Interaction with `SsTableView`" section noting that `SstReader::open_with_handle()` accepts `SsTableHandle` via `view.sst`. Refined estimate section updated to account for view-level `visible_range` projections on boundary SSTs. \ No newline at end of file +- **2026-03-19**: Updated RFC to reflect `SsTableView` indirection introduced in [#1362](https://github.com/slatedb/slatedb/pull/1362). `ManifestCore` and `SortedRun` now contain `SsTableView` references instead of raw `SsTableHandle`s. Updated `tables_covering_range()` return type to `VecDeque<&SsTableView>`, replaced `SsTableHandle::estimate_size()`/`visible_range()` references with `SsTableView` equivalents, and added "Interaction with `SsTableView`" section noting that `SstReader::open_with_handle()` accepts `SsTableHandle` via `view.sst`. Refined estimate section updated to account for view-level `visible_range` projections on boundary SSTs. +- **2026-08-11**: Changed the return type of `SstFile::index` from `Vec<(u64, Bytes)>` to `SstIndex`. `SstIndex` is an opaque view into the underlying FlatBuffer via `SsTableIndexOwned`. `SstIndex` exposes block offsets and first keys via an `ExactSizeIterator` and lets the user search for a block via a `partition_point` method that uses the SlateDB internal binary search for finding a block offset. This is a breaking change as the return type of `SstFile::index` changes. diff --git a/slatedb/src/lib.rs b/slatedb/src/lib.rs index f9d05befe2..92131c15d3 100644 --- a/slatedb/src/lib.rs +++ b/slatedb/src/lib.rs @@ -72,7 +72,7 @@ pub use prefix_extractor::{PrefixExtractor, PrefixTarget}; pub use slatedb_common::{DbRand, IdentifiedObjectMetadata, ObjectMetadata}; #[cfg(test)] pub use sst_builder::BlockFormat; -pub use sst_reader::{SstFile, SstReader}; +pub use sst_reader::{SstFile, SstIndex, SstReader}; pub use sst_stats::{BlockStats, SstStats}; pub use transaction_manager::IsolationLevel; pub use types::KeyValue; diff --git a/slatedb/src/partitioned_keyspace.rs b/slatedb/src/partitioned_keyspace.rs index 69e6911166..2da8bb0798 100644 --- a/slatedb/src/partitioned_keyspace.rs +++ b/slatedb/src/partitioned_keyspace.rs @@ -13,7 +13,7 @@ pub(crate) trait RangePartitionedKeySpace { } // equivalent to https://doc.rust-lang.org/std/primitive.slice.html#method.partition_point -fn partition_point bool>( +pub(crate) fn partition_point bool>( keyspace: &T, pred: P, ) -> usize { diff --git a/slatedb/src/sst_reader.rs b/slatedb/src/sst_reader.rs index 6d852fe1b5..c880d5e719 100644 --- a/slatedb/src/sst_reader.rs +++ b/slatedb/src/sst_reader.rs @@ -47,7 +47,6 @@ use std::sync::Arc; -use bytes::Bytes; use object_store::path::Path; use object_store::ObjectStore; use ulid::Ulid; @@ -56,9 +55,11 @@ use crate::block_cache_policy::BlockCachePolicy; use crate::block_iterator::DataBlockIterator; use crate::db_cache::DbCache; use crate::db_state::{SsTableHandle, SsTableId, SsTableInfo}; +use crate::flatbuffer_types::SsTableIndexOwned; use crate::format::sst::{BlockTransformer, SsTableFormat}; use crate::iter::IterationOrder; use crate::object_stores::ObjectStores; +use crate::partitioned_keyspace::{partition_point, RangePartitionedKeySpace}; use crate::sst_stats::SstStats; use crate::tablestore::{TableStore, TableStoreKind}; use crate::types::RowEntry; @@ -147,6 +148,69 @@ pub struct SstFile { table_store: Arc, } +/// A zero-copy view of an SST's block index. +/// +/// The view owns a reference to the cached index data. Keys returned by its +/// accessors borrow directly from that data without allocation or copying. +#[derive(Clone)] +pub struct SstIndex { + inner: Arc, +} + +impl SstIndex { + /// Returns the number of data blocks described by the index. + pub fn len(&self) -> usize { + self.inner.borrow().block_meta().len() + } + + /// Returns whether the index contains no data blocks. + pub fn is_empty(&self) -> bool { + self.len() == 0 + } + + /// Returns the block offset and first key at `index`. + pub fn get(&self, index: usize) -> Option<(u64, &[u8])> { + let block_meta = self.inner.borrow().block_meta(); + if index >= block_meta.len() { + return None; + } + let meta = block_meta.get(index); + Some((meta.offset(), meta.first_key().bytes())) + } + + /// Iterates over block offsets and first keys in index order. + pub fn iter(&self) -> impl ExactSizeIterator + '_ { + let block_meta = self.inner.borrow().block_meta(); + (0..block_meta.len()).map(move |index| { + let meta = block_meta.get(index); + (meta.offset(), meta.first_key().bytes()) + }) + } + + /// Returns the first index for which `pred` is false. + /// + /// The index first keys are sorted, so `pred` must return `true` for a + /// contiguous prefix of the index, matching [`slice::partition_point`]. + pub fn partition_point

(&self, pred: P) -> usize + where + P: Fn(&[u8]) -> bool, + { + partition_point(self, pred) + } +} + +impl RangePartitionedKeySpace for SstIndex { + fn partitions(&self) -> usize { + self.len() + } + + fn partition_first_key(&self, partition: usize) -> &[u8] { + self.get(partition) + .expect("partition index should be in range") + .1 + } +} + impl SstFile { /// Returns the SST's ULID identifier. pub fn id(&self) -> Ulid { @@ -192,29 +256,20 @@ impl SstFile { .map_err(Into::into) } - /// Returns `(block_offset, first_key)` pairs from the SST index block. + /// Returns a zero-copy view of the SST index block. /// - /// The returned vector is parallel to the data blocks in the SST. Each - /// entry contains the on-disk byte offset of the block and the first key - /// stored in that block. + /// The returned [`SstIndex`] contains one entry for each data block in the + /// SST, in block order. Each entry contains the on-disk byte offset of + /// the block and the first key stored in that block. The index keeps + /// the cached index data alive, and keys returned by its accessors borrow + /// directly from that data without allocation or copying. /// /// ## Errors /// /// Returns an error if there is an issue reading from object storage. - pub async fn index(&self) -> Result, crate::Error> { - let index = self.table_store.read_index(&self.handle, true).await?; - let borrowed = index.borrow(); - let block_meta = borrowed.block_meta(); - let result: Vec<(u64, Bytes)> = (0..block_meta.len()) - .map(|i| { - let meta = block_meta.get(i); - ( - meta.offset(), - Bytes::copy_from_slice(meta.first_key().bytes()), - ) - }) - .collect(); - Ok(result) + pub async fn index(&self) -> Result { + let inner = self.table_store.read_index(&self.handle, true).await?; + Ok(SstIndex { inner }) } /// Reads a single data block by its index and returns the decoded rows. @@ -262,6 +317,7 @@ mod tests { use crate::test_utils::StringConcatMergeOperator; use crate::types::ValueDeletable; use crate::Db; + use bytes::Bytes; use object_store::memory::InMemory; /// Helper: create a DB with 10 puts, 3 deletes, and 2 merges, flush to @@ -376,11 +432,41 @@ mod tests { // First index key should be <= the SST's first entry (it may be a // shortened separator key rather than the exact first key). if let Some(first_entry) = sst_file.info().first_entry.as_ref() { - assert!(index[0].1.as_ref() <= first_entry.as_ref()); + let (_, first_key) = index.get(0).expect("index should not be empty"); + assert!(first_key <= first_entry.as_ref()); } // Offsets should be monotonically increasing - for window in index.windows(2) { - assert!(window[0].0 < window[1].0); + let mut entries = index.iter(); + let mut previous_offset = entries.next().expect("index should not be empty").0; + for (offset, _) in entries { + assert!(previous_offset < offset); + previous_offset = offset; + } + assert_eq!(index.get(index.len()), None); + } + + #[tokio::test] + async fn test_index_partition_point_matches_slice() { + let (store, path, manifest) = setup_db_with_l0().await; + let reader = SstReader::new(path, store, None, None); + + let view = &manifest.manifest.core.tree.l0[0]; + let sst_file = reader.open_with_handle(view.sst.clone()).unwrap(); + let index = sst_file.index().await.unwrap(); + let owned = index + .iter() + .map(|(offset, first_key)| (offset, Bytes::copy_from_slice(first_key))) + .collect::>(); + + for key in [b"".as_slice(), b"k00", b"k05", b"k99"] { + assert_eq!( + index.partition_point(|candidate| candidate < key), + owned.partition_point(|(_, candidate)| candidate.as_ref() < key) + ); + assert_eq!( + index.partition_point(|candidate| candidate <= key), + owned.partition_point(|(_, candidate)| candidate.as_ref() <= key) + ); } } From e9a14cd614b4200d9604fbcbb362d5c395cccd0a Mon Sep 17 00:00:00 2001 From: Rohan Date: Tue, 11 Aug 2026 22:06:00 -0400 Subject: [PATCH 30/65] [rfc-30 6/N]: wire custom WAL into builders (#2015) --- rfcs/0030-pluggable-wal.md | 5 + slatedb/src/admin.rs | 46 ++- slatedb/src/clone.rs | 16 +- slatedb/src/db/builder.rs | 137 ++++++- slatedb/src/db_reader.rs | 181 +++++--- slatedb/src/fence.rs | 41 +- slatedb/src/garbage_collector.rs | 42 +- slatedb/src/garbage_collector/wal_gc.rs | 22 +- slatedb/src/wal/admin.rs | 25 +- slatedb/src/wal/gc.rs | 93 ++--- slatedb/src/wal/mod.rs | 28 +- slatedb/src/wal/reader.rs | 114 ++++++ slatedb/src/wal_replay.rs | 1 + slatedb/tests/custom_wal.rs | 524 ++++++++++++++++++++++++ 14 files changed, 1097 insertions(+), 178 deletions(-) create mode 100644 slatedb/src/wal/reader.rs create mode 100644 slatedb/tests/custom_wal.rs diff --git a/rfcs/0030-pluggable-wal.md b/rfcs/0030-pluggable-wal.md index 3886aae20a..245b051c26 100644 --- a/rfcs/0030-pluggable-wal.md +++ b/rfcs/0030-pluggable-wal.md @@ -389,6 +389,11 @@ pub trait WalReader { &self, wal_file_id_range: WalFileRange, ) -> Result, WalError>; + + /// Returns the ID of the last WAL file currently present after `replay_after_wal_id`, or + /// `replay_after_wal_id` if no later WAL file is present. Implementations may use + /// `replay_after_wal_id` as a known lower bound when locating the end of the WAL. + async fn last_wal_file_id(&self, replay_after_wal_id: u64) -> Result; } /// API for plugging into WAL GC diff --git a/slatedb/src/admin.rs b/slatedb/src/admin.rs index 38e322f6f6..49ea0182bb 100644 --- a/slatedb/src/admin.rs +++ b/slatedb/src/admin.rs @@ -1,5 +1,6 @@ pub use crate::db::builder::CloneBuilder; pub use crate::db::builder::CloneSourceSpec; +use std::collections::BTreeSet; use crate::checkpoint::{Checkpoint, CheckpointCreateResult}; use crate::compactions_store::CompactionsStore; @@ -35,6 +36,7 @@ use uuid::Uuid; pub use crate::db::builder::AdminBuilder; use crate::merge_operator::MergeOperatorType; +use crate::wal::WalAdmin; use slatedb_txn_obj::TransactionalObject; /// An Admin struct for SlateDB administration operations. @@ -57,6 +59,7 @@ pub struct Admin { pub(crate) compaction_filter_supplier: Option>, pub(crate) merge_operator: Option, + pub(crate) wal_admin: Arc, } impl Admin { @@ -284,6 +287,7 @@ impl Admin { self.object_stores.store_of(ObjectStoreType::Main).clone(), ) .with_system_clock(self.system_clock.clone()) + .with_wal_gc(self.wal_admin.garbage_collector(&self.path)) .with_wal_object_store(self.object_stores.store_of(ObjectStoreType::Wal).clone()) .with_options(gc_opts) .with_seed(self.rand.rng().next_u64()) @@ -317,6 +321,7 @@ impl Admin { self.object_stores.store_of(ObjectStoreType::Main).clone(), ) .with_system_clock(self.system_clock.clone()) + .with_wal_gc(self.wal_admin.garbage_collector(&self.path)) .with_wal_object_store(self.object_stores.store_of(ObjectStoreType::Wal).clone()) .with_options(gc_opts) .with_seed(self.rand.rng().next_u64()) @@ -573,7 +578,7 @@ impl Admin { /// job. A `confirm` delete of a dir with neither a manifest nor a marker is /// refused, so a fat-fingered `--path` can't wipe an unrelated directory. /// Idempotent. - pub async fn delete_db(&self, confirm: bool) -> Result, crate::Error> { + pub async fn delete_db(&self, confirm: bool) -> Result, crate::Error> { let main = self.retrying_store(ObjectStoreType::Main); if !confirm { @@ -631,26 +636,31 @@ impl Admin { // Delete everything but the marker, then the marker last, so a crash in // between leaves the marker to prove a rerun should finish. let mut deleted = self.delete_prefix(&main, Some(&marker)).await?; - if self.object_stores.has_wal_object_store() { - deleted.extend( - self.delete_prefix(&self.retrying_store(ObjectStoreType::Wal), None) - .await?, - ); - } + deleted.extend( + self.wal_admin + .delete_wal(&self.path, false) + .await + .map_err(SlateDBError::from)?, + ); main.delete(&marker).await.map_err(SlateDBError::from)?; - deleted.push(marker); + deleted.push(marker.to_string()); Ok(deleted) } /// Lists every object under this db's path prefix across the main and WAL stores. - async fn list_prefix(&self, main: &Arc) -> Result, crate::Error> { - let mut paths = collect_prefix(main, &self.path).await?; - if self.object_stores.has_wal_object_store() { - paths.extend( - collect_prefix(&self.retrying_store(ObjectStoreType::Wal), &self.path).await?, - ); + async fn list_prefix(&self, main: &Arc) -> Result, crate::Error> { + let paths = collect_prefix(main, &self.path).await?; + // track the dry run paths in a set since the WAL and db may overlap + let mut paths = paths.iter().map(Path::to_string).collect::>(); + let wal_paths = self + .wal_admin + .delete_wal(&self.path, true) + .await + .map_err(SlateDBError::from)?; + for wp in wal_paths { + paths.insert(wp); } - Ok(paths) + Ok(paths.into_iter().collect()) } /// Deletes every object under this db's path prefix in the given store, @@ -659,7 +669,7 @@ impl Admin { &self, store: &Arc, keep: Option<&Path>, - ) -> Result, crate::Error> { + ) -> Result, crate::Error> { let mut deleted = Vec::new(); for path in collect_prefix(store, &self.path).await? { if Some(&path) == keep { @@ -668,7 +678,7 @@ impl Admin { store.delete(&path).await.map_err(SlateDBError::from)?; deleted.push(path); } - Ok(deleted) + Ok(deleted.into_iter().map(|p| p.to_string()).collect()) } /// Returns the timestamp or sequence from the latest manifest's sequence tracker. @@ -782,7 +792,7 @@ impl Admin { source, self.retrying_store(ObjectStoreType::Main), ) - .with_wal_object_store(self.retrying_store(ObjectStoreType::Wal)) + .with_wal_admin(self.wal_admin.clone()) } /// Creates a new builder for an admin client at the given path. diff --git a/slatedb/src/clone.rs b/slatedb/src/clone.rs index 6e40e0bb30..d5092c88ca 100644 --- a/slatedb/src/clone.rs +++ b/slatedb/src/clone.rs @@ -622,6 +622,7 @@ mod tests { use std::ops::Bound; use std::ops::RangeBounds; use std::sync::Arc; + use std::time::Duration; use uuid::Uuid; struct RemappingWalAdmin { @@ -633,19 +634,24 @@ mod tests { #[async_trait] impl WalGc for NoopWalGc { - async fn collect(&self, _referenced_ranges: Vec) -> Result<(), WalError> { + async fn collect( + &self, + _referenced_ranges: Vec, + _min_age: Duration, + _dry_run: bool, + ) -> Result<(), WalError> { Ok(()) } } #[async_trait] impl WalAdmin for RemappingWalAdmin { - fn garbage_collector(&self, _path: &Path) -> Box { - Box::new(NoopWalGc) + fn garbage_collector(&self, _path: &Path) -> Arc { + Arc::new(NoopWalGc) } - async fn delete_wal(&self, _path: &Path) -> Result<(), WalError> { - Ok(()) + async fn delete_wal(&self, _path: &Path, _dry_run: bool) -> Result, WalError> { + Ok(vec![]) } async fn is_empty( diff --git a/slatedb/src/db/builder.rs b/slatedb/src/db/builder.rs index ffb103d814..22192e0899 100644 --- a/slatedb/src/db/builder.rs +++ b/slatedb/src/db/builder.rs @@ -161,9 +161,10 @@ use crate::retrying_object_store::RetryingObjectStore; use crate::tablestore::{TableStore, TableStoreKind}; use crate::utils::SafeSender; use crate::utils::WatchableOnceCell; +use crate::wal; use crate::wal::admin::SlateDbWalAdmin; use crate::wal::wal_disabled::DisabledWalObserver; -use crate::wal::WalObserver; +use crate::wal::{WalAdmin, WalGc, WalObserver}; use slatedb_common::clock::DefaultSystemClock; use slatedb_common::clock::SystemClock; use slatedb_common::metrics::MetricsRecorder; @@ -195,6 +196,7 @@ pub struct DbBuilder> { filter_policies: Vec>, metrics_recorder: Arc, segment_extractor: Option>, + wal_writer_init: Option>, } impl> DbBuilder

{ @@ -224,6 +226,7 @@ impl> DbBuilder

{ filter_policies: default_filter_policies(), metrics_recorder: Arc::new(NoopMetricsRecorder::new()), segment_extractor: None, + wal_writer_init: None, } } @@ -265,6 +268,14 @@ impl> DbBuilder

{ self } + /// Sets the WAL writer initializer used to fence the WAL and create the writer. + /// Use this to plug in a custom WAL implementation. SlateDB's object-store-backed + /// WAL is used by default. + pub fn with_wal_writer(mut self, wal_writer_init: Box) -> Self { + self.wal_writer_init = Some(wal_writer_init); + self + } + /// Sets the cache to use for the database for caching sst blocks /// /// SlateDB uses a cache to efficiently store and retrieve blocks and SST metadata locally. @@ -615,6 +626,7 @@ impl> DbBuilder

{ &self.settings, system_clock.clone(), task_executor.clone(), + self.wal_writer_init, ); let WriterFenceResult { manifest, @@ -843,6 +855,7 @@ pub struct AdminBuilder> { path: P, main_object_store: Arc, wal_object_store: Option>, + wal_admin: Option>, system_clock: Arc, rand: Arc, object_store_max_retries: Option, @@ -858,6 +871,7 @@ impl> AdminBuilder

{ path, main_object_store, wal_object_store: None, + wal_admin: None, system_clock: Arc::new(DefaultSystemClock::new()), rand: Arc::new(DbRand::default()), object_store_max_retries: None, @@ -883,6 +897,13 @@ impl> AdminBuilder

{ self } + /// Sets the WAL admin API implementation to use for admin operations like db deletion + /// and cloning. Users should set this if using a custom WAL implementation. + pub fn with_wal_admin(mut self, wal_admin: Arc) -> Self { + self.wal_admin = Some(wal_admin); + self + } + /// Sets the random number generator to use for randomness. pub fn with_seed(mut self, seed: u64) -> Self { self.rand = Arc::new(DbRand::new(seed)); @@ -915,9 +936,23 @@ impl> AdminBuilder

{ // rather than at build time, because several admin operations delegate // to sub-builders (compactor/GC) that add their own retry layer, and // wrapping here would double-wrap them. + let object_stores = ObjectStores::new(self.main_object_store, self.wal_object_store); + let wal_admin = self.wal_admin.unwrap_or_else(|| { + let retrying_object_store = Arc::new(RetryingObjectStore::new( + object_stores.store_of(ObjectStoreType::Wal).clone(), + self.rand.clone(), + self.system_clock.clone(), + self.object_store_max_retries, + )); + Arc::new(SlateDbWalAdmin::new( + retrying_object_store, + Arc::new(FailPointRegistry::new()), + )) + }); Admin { path: self.path.into(), - object_stores: ObjectStores::new(self.main_object_store, self.wal_object_store), + object_stores, + wal_admin, system_clock: self.system_clock, rand: self.rand, object_store_max_retries: self.object_store_max_retries, @@ -935,6 +970,7 @@ pub struct GarbageCollectorBuilder> { path: P, main_object_store: Arc, wal_object_store: Option>, + wal_gc: Option>, options: GarbageCollectorOptions, gc_filter: Option>, metrics_recorder: Arc, @@ -948,6 +984,7 @@ impl> GarbageCollectorBuilder

{ path, main_object_store, wal_object_store: None, + wal_gc: None, options: GarbageCollectorOptions::default(), gc_filter: None, metrics_recorder: Arc::new(NoopMetricsRecorder::new()), @@ -962,6 +999,7 @@ impl> GarbageCollectorBuilder

{ path: self.path.into(), main_object_store: self.main_object_store, wal_object_store: self.wal_object_store, + wal_gc: self.wal_gc, options: self.options, gc_filter: self.gc_filter, metrics_recorder: self.metrics_recorder, @@ -1009,6 +1047,11 @@ impl> GarbageCollectorBuilder

{ self } + pub fn with_wal_gc(mut self, wal_gc: Arc) -> Self { + self.wal_gc = Some(wal_gc); + self + } + /// Builds a GarbageCollector using pre-existing stores (used by DbBuilder). pub(crate) fn build_collector( self, @@ -1030,6 +1073,7 @@ impl> GarbageCollectorBuilder

{ &recorder, self.system_clock, self.gc_filter, + self.wal_gc, ) } @@ -1088,6 +1132,7 @@ impl> GarbageCollectorBuilder

{ &recorder, self.system_clock, self.gc_filter, + self.wal_gc, ) } } @@ -1617,6 +1662,7 @@ pub struct DbReaderBuilder> { path: P, object_store: Arc, wal_object_store: Option>, + wal_reader: Option>, db_cache: Option>, mode: DbReaderMode, merge_operator: Option, @@ -1636,6 +1682,7 @@ impl> DbReaderBuilder

{ path, object_store, wal_object_store: None, + wal_reader: None, db_cache: default_db_cache(), mode: DbReaderMode::default(), merge_operator: None, @@ -1662,6 +1709,13 @@ impl> DbReaderBuilder

{ self } + /// Sets the WAL reader used to replay WAL rows. Use this to plug in a custom + /// WAL implementation. SlateDB's object-store-backed WAL is used by default. + pub fn with_wal_reader(mut self, wal_reader: Arc) -> Self { + self.wal_reader = Some(wal_reader); + self + } + /// Sets the merge operator to use when reading merge operands. pub fn with_merge_operator(mut self, merge_operator: MergeOperatorType) -> Self { self.merge_operator = Some(merge_operator); @@ -1868,6 +1922,7 @@ impl> DbReaderBuilder

{ manifest_store, table_store, self.mode, + self.wal_reader, self.merge_operator, self.segment_extractor, self.options, @@ -1955,6 +2010,7 @@ pub struct CloneBuilder + Clone = (Bound, Bound>, object_store: Arc, wal_object_store: Option>, + wal_admin: Option>, system_clock: Option>, rand: Option>, projection_range: Option, @@ -1973,6 +2029,7 @@ impl + Clone> CloneBuilder { sources: vec![source], object_store, wal_object_store: None, + wal_admin: None, system_clock: None, rand: None, projection_range: None, @@ -2001,6 +2058,11 @@ impl + Clone> CloneBuilder { self } + pub fn with_wal_admin(mut self, wal_admin: Arc) -> Self { + self.wal_admin = Some(wal_admin); + self + } + pub fn with_projection_range(mut self, projection_range: Option) -> Self { self.projection_range = projection_range; self @@ -2056,10 +2118,14 @@ impl + Clone> CloneBuilder { /// Build and execute the clone operation. pub async fn build(self) -> Result<(), crate::Error> { let fp_registry = Arc::new(FailPointRegistry::new()); - let wal_admin = Arc::new(SlateDbWalAdmin::new( - self.wal_object_store.unwrap_or(self.object_store.clone()), - fp_registry.clone(), - )); + let wal_admin = if let Some(wal_admin) = self.wal_admin { + wal_admin + } else { + Arc::new(SlateDbWalAdmin::new( + self.wal_object_store.unwrap_or(self.object_store.clone()), + fp_registry.clone(), + )) + }; crate::clone::create_clone( self.sources, self.clone_path, @@ -2159,6 +2225,10 @@ mod tests { use crate::instrumented_object_store::stats::REQUEST_COUNT as OBJECT_STORE_REQUEST_COUNT; use crate::manifest::store::{ManifestStore, StoredManifest}; use crate::manifest::ManifestCore; + use crate::wal::test_utils::FakeWalWriter; + use crate::wal::{ + WalError, WalIterator, WalRows, WriterInit, WriterInitResult, WriterManifest, + }; use object_store::memory::InMemory; use object_store::path::Path; use object_store::ObjectStore; @@ -2166,7 +2236,37 @@ mod tests { use slatedb_common::metrics::{ lookup_metric, lookup_metric_with_labels, DefaultMetricsRecorder, MetricsRecorderHelper, }; - use std::sync::Arc; + use std::sync::{ + atomic::{AtomicBool, Ordering}, + Arc, + }; + + struct EmptyWalIterator; + + #[async_trait::async_trait] + impl WalIterator for EmptyWalIterator { + async fn next(&mut self) -> Result, WalError> { + Ok(None) + } + } + + struct RecordingWriterInit { + called: Arc, + } + + #[async_trait::async_trait] + impl WriterInit for RecordingWriterInit { + async fn fence_and_init( + &self, + manifest: &mut WriterManifest, + ) -> Result { + self.called.store(true, Ordering::Relaxed); + Ok(WriterInitResult { + replay_iterator: Box::new(EmptyWalIterator), + wal_writer: Box::new(FakeWalWriter::new(manifest.replay_after_wal_id())), + }) + } + } fn object_store_labels( component: &'static str, @@ -2192,6 +2292,29 @@ mod tests { assert_eq!(lookup_metric(recorder, name), Some(1)); } + #[tokio::test] + async fn test_db_builder_uses_custom_wal_writer_init() { + let called = Arc::new(AtomicBool::new(false)); + let db = crate::Db::builder( + "test_db_builder_uses_custom_wal_writer_init", + Arc::new(InMemory::new()), + ) + .with_settings(Settings { + compactor_options: None, + garbage_collector_options: None, + ..Settings::default() + }) + .with_wal_writer(Box::new(RecordingWriterInit { + called: Arc::clone(&called), + })) + .build() + .await + .expect("failed to build db"); + + assert!(called.load(Ordering::Relaxed)); + db.close().await.expect("failed to close db"); + } + #[tokio::test] async fn test_db_builder_starts_gc_by_default() { let metrics_recorder = Arc::new(DefaultMetricsRecorder::new()); diff --git a/slatedb/src/db_reader.rs b/slatedb/src/db_reader.rs index 8b3b05a5aa..badc554d78 100644 --- a/slatedb/src/db_reader.rs +++ b/slatedb/src/db_reader.rs @@ -12,7 +12,6 @@ use { db_status::{ClosedResultWriter, DbStatus, DbStatusManager}, dispatcher::{MessageHandler, MessageHandlerExecutor, MessageTickerDef}, error::SlateDBError, - iter::IterationOrder, manifest::{ store::{ManifestStore, StoredManifest}, Manifest, ManifestCore, VersionedManifest, @@ -23,11 +22,11 @@ use { paths::PathResolver, prefix_extractor::PrefixExtractor, reader::{DbStateReader, Reader, ScanContext}, - sst_iter::SstIteratorOptions, tablestore::TableStore, types::KeyValue, utils::IdGenerator, - wal_replay::{WalIteratorOptions, WalReplayIterator, WalReplayOptions}, + wal::WalReader as WalReaderTrait, + wal_replay::{WalReplayIterator, WalReplayOptions}, Checkpoint, DbCacheManagerOps, DbIterator, DbMetadataOps, DbReadOps, }, async_trait::async_trait, @@ -83,7 +82,7 @@ enum WalReplayEnd { /// the manifest itself records as durable. Manifest, - /// Probe the object store for the newest WAL file and replay through it, + /// Ask the configured WAL reader for the newest WAL file and replay through it, /// picking up writes made after the manifest was written. Latest, } @@ -115,6 +114,7 @@ pub struct DbReader { struct DbReaderInner { manifest_store: Arc, table_store: Arc, + wal_reader: Arc, options: DbReaderOptions, mode: DbReaderMode, state: RwLock>, @@ -172,6 +172,7 @@ impl DbReaderInner { async fn new( manifest_store: Arc, table_store: Arc, + wal_reader: Option>, options: DbReaderOptions, mode: DbReaderMode, merge_operator: Option, @@ -181,6 +182,11 @@ impl DbReaderInner { recorder: slatedb_common::metrics::MetricsRecorderHelper, mut manifest: StoredManifest, ) -> Result { + let wal_reader = wal_reader.unwrap_or_else(|| { + Arc::new(crate::wal::reader::SlateDbWalReader::new(Arc::clone( + &table_store, + ))) + }); let checkpoint = Self::get_or_create_checkpoint(&mut manifest, mode, &options, rand.clone()).await?; let (manifest_id, initial_manifest) = if let Some(checkpoint) = checkpoint.as_ref() { @@ -199,6 +205,7 @@ impl DbReaderInner { VecDeque::new(), WalReplayEnd::for_reader(mode, &options), Arc::clone(&table_store), + wal_reader.as_ref(), &options, segment_extractor.as_ref(), ) @@ -239,6 +246,7 @@ impl DbReaderInner { let inner = Self { manifest_store, table_store, + wal_reader, options, mode, state, @@ -384,25 +392,20 @@ impl DbReaderInner { if self.options.skip_wal_replay { return Ok(()); } - let last_replayed_wal_id = self.state.read().last_wal_id; - let last_seen_wal_id = self - .table_store - .last_seen_wal_id(last_replayed_wal_id) - .await?; - if last_seen_wal_id > last_replayed_wal_id { - let current_state = Arc::clone(&self.state.read()); - let mut imm_memtable = current_state.imm_memtable().clone(); - - let (last_wal_id, last_committed_seq) = Self::replay_wal_into( - Arc::clone(&self.table_store), - &self.options, - current_state.core(), - &mut imm_memtable, - WalReplayEnd::Latest, - self.segment_extractor.as_ref(), - ) - .await?; + let current_state = Arc::clone(&self.state.read()); + let mut imm_memtable = current_state.imm_memtable().clone(); + let (last_wal_id, last_committed_seq) = Self::replay_wal_into( + Arc::clone(&self.table_store), + self.wal_reader.as_ref(), + &self.options, + current_state.core(), + &mut imm_memtable, + WalReplayEnd::Latest, + self.segment_extractor.as_ref(), + ) + .await?; + if last_wal_id > current_state.last_wal_id { self.oracle.advance_durable_seq(last_committed_seq); let mut write_guard = self.state.write(); *write_guard = Arc::new(ReaderState { @@ -470,6 +473,7 @@ impl DbReaderInner { imm_memtable, WalReplayEnd::for_reader(self.mode, &self.options), Arc::clone(&self.table_store), + self.wal_reader.as_ref(), &self.options, self.segment_extractor.as_ref(), ) @@ -483,6 +487,7 @@ impl DbReaderInner { mut imm_memtable: VecDeque>, replay_wals: Option, table_store: Arc, + wal_reader: &dyn WalReaderTrait, options: &DbReaderOptions, segment_extractor: Option<&Arc>, ) -> Result { @@ -490,6 +495,7 @@ impl DbReaderInner { Some(replay_end) => { Self::replay_wal_into( Arc::clone(&table_store), + wal_reader, options, &manifest.core, &mut imm_memtable, @@ -637,34 +643,30 @@ impl DbReaderInner { async fn replay_wal_into( table_store: Arc, + wal_reader: &dyn WalReaderTrait, reader_options: &DbReaderOptions, core: &ManifestCore, into_tables: &mut VecDeque>, replay_end: WalReplayEnd, segment_extractor: Option<&Arc>, ) -> Result<(u64, u64), SlateDBError> { - let sst_iter_options = SstIteratorOptions { - max_fetch_tasks: 1, - blocks_to_fetch: 256, - cache_blocks: true, - cache_metadata: false, - eager_spawn: true, - order: IterationOrder::Ascending, - prefix: None, - filter_context: None, - }; - let (mut replay_after_wal_id, mut last_committed_seq) = Self::replayed_watermark(core, into_tables); + let wal_id_start = replay_after_wal_id + .checked_add(1) + .ok_or(SlateDBError::InvalidDBState)?; let wal_id_end = match replay_end { WalReplayEnd::Manifest => core.next_wal_sst_id, - WalReplayEnd::Latest => table_store.last_seen_wal_id(replay_after_wal_id).await? + 1, + WalReplayEnd::Latest => wal_reader + .last_wal_file_id(replay_after_wal_id) + .await? + .checked_add(1) + .ok_or(SlateDBError::InvalidDBState)?, }; + if wal_id_start >= wal_id_end { + return Ok((replay_after_wal_id, last_committed_seq)); + } - let iterator_options = WalIteratorOptions { - sst_batch_size: 4, - sst_iter_options, - }; let replay_options = WalReplayOptions { max_memtable_bytes: reader_options.max_memtable_bytes as usize, // Skip entries that we already have in `imm_memtable` (that might be above @@ -672,10 +674,12 @@ impl DbReaderInner { min_seq: Some(last_committed_seq), }; - let mut replay_iter = WalReplayIterator::range( - (replay_after_wal_id + 1)..wal_id_end, + let wal_iter = wal_reader + .iterator((wal_id_start..wal_id_end).into()) + .await?; + let mut replay_iter = WalReplayIterator::for_wal_iterator( + wal_iter, core, - iterator_options, replay_options, Arc::clone(&table_store), )?; @@ -945,6 +949,7 @@ impl DbReader { manifest_store: Arc, table_store: Arc, mode: DbReaderMode, + wal_reader: Option>, merge_operator: Option, segment_extractor: Option>, options: DbReaderOptions, @@ -968,6 +973,7 @@ impl DbReader { DbReaderInner::new( manifest_store, table_store, + wal_reader, options, mode, merge_operator, @@ -1483,6 +1489,7 @@ mod tests { tablestore::{TableStore, TableStoreKind}, test_utils, types::RowEntry, + wal::{WalError, WalFileRange, WalIterator, WalReader as WalReaderTrait, WalRows}, CloseReason, Db, }, bytes::Bytes, @@ -1495,12 +1502,46 @@ mod tests { }, std::{ collections::{BTreeMap, VecDeque}, - sync::Arc, + sync::{ + atomic::{AtomicUsize, Ordering}, + Arc, + }, time::Duration, }, uuid::Uuid, }; + struct EmptyTestWalIterator; + + #[async_trait::async_trait] + impl WalIterator for EmptyTestWalIterator { + async fn next(&mut self) -> Result, WalError> { + Ok(None) + } + } + + #[derive(Default)] + struct CountingWalReader { + iterator_calls: AtomicUsize, + last_wal_file_id_calls: AtomicUsize, + } + + #[async_trait::async_trait] + impl WalReaderTrait for CountingWalReader { + async fn iterator( + &self, + _wal_file_id_range: WalFileRange, + ) -> Result, WalError> { + self.iterator_calls.fetch_add(1, Ordering::Relaxed); + Ok(Box::new(EmptyTestWalIterator)) + } + + async fn last_wal_file_id(&self, _replay_after_wal_id: u64) -> Result { + self.last_wal_file_id_calls.fetch_add(1, Ordering::Relaxed); + Ok(10) + } + } + #[tokio::test] async fn should_get_value_from_db() { let object_store: Arc = Arc::new(InMemory::new()); @@ -1534,6 +1575,32 @@ mod tests { ); } + #[tokio::test] + async fn db_reader_builder_should_use_custom_wal_reader() { + let object_store: Arc = Arc::new(InMemory::new()); + let path = Path::from("/tmp/test_custom_wal_reader"); + let db = Db::open(path.clone(), Arc::clone(&object_store)) + .await + .unwrap(); + db.close().await.unwrap(); + + let wal_reader = Arc::new(CountingWalReader::default()); + let reader = DbReader::builder(path, object_store) + .with_reader_mode(DbReaderMode::FollowLatest) + .with_wal_reader(wal_reader.clone()) + .with_options(DbReaderOptions { + manifest_poll_interval: Duration::from_secs(60 * 60), + ..DbReaderOptions::default() + }) + .build() + .await + .unwrap(); + + assert!(wal_reader.last_wal_file_id_calls.load(Ordering::Relaxed) > 0); + assert!(wal_reader.iterator_calls.load(Ordering::Relaxed) > 0); + reader.close().await.unwrap(); + } + #[tokio::test] async fn should_return_current_versioned_manifest() { let object_store: Arc = Arc::new(InMemory::new()); @@ -1584,6 +1651,7 @@ mod tests { DbReaderMode::Checkpoint(checkpoint_result.id), None, None, + None, DbReaderOptions::default(), test_provider.system_clock.clone(), test_provider.rand.clone(), @@ -1943,6 +2011,7 @@ mod tests { DbReaderMode::FollowLatest, None, None, + None, DbReaderOptions { manifest_poll_interval: Duration::from_secs(60 * 60), ..DbReaderOptions::default() @@ -2136,6 +2205,7 @@ mod tests { let inner = DbReaderInner::new( Arc::clone(&manifest_store), table_store, + None, DbReaderOptions { manifest_poll_interval: Duration::from_millis(100), checkpoint_lifetime: Duration::from_millis(1000), @@ -2231,6 +2301,7 @@ mod tests { let inner = DbReaderInner::new( Arc::clone(&manifest_store), table_store, + None, DbReaderOptions { manifest_poll_interval: Duration::from_millis(100), checkpoint_lifetime: Duration::from_millis(1000), @@ -2347,6 +2418,7 @@ mod tests { let (last_wal_id, last_committed_seq) = DbReaderInner::replay_wal_into( Arc::clone(&table_store), + &native_wal_reader(&table_store), &DbReaderOptions::default(), &core, &mut into_tables, @@ -2392,6 +2464,7 @@ mod tests { let (last_wal_id, last_committed_seq) = DbReaderInner::replay_wal_into( Arc::clone(&table_store), + &native_wal_reader(&table_store), &DbReaderOptions::default(), &core, &mut into_tables, @@ -2447,6 +2520,7 @@ mod tests { let (last_wal_id, last_committed_seq) = DbReaderInner::replay_wal_into( Arc::clone(&table_store), + &native_wal_reader(&table_store), &reader_options, &core, &mut into_tables, @@ -2483,6 +2557,7 @@ mod tests { let (last_wal_id, last_committed_seq) = DbReaderInner::replay_wal_into( Arc::clone(&table_store), + &native_wal_reader(&table_store), &DbReaderOptions::default(), &core, &mut into_tables, @@ -2514,6 +2589,7 @@ mod tests { let (last_wal_id, last_committed_seq) = DbReaderInner::replay_wal_into( Arc::clone(&table_store), + &native_wal_reader(&table_store), &DbReaderOptions::default(), &core, &mut into_tables, @@ -2560,6 +2636,7 @@ mod tests { let (last_wal_id, last_committed_seq) = DbReaderInner::replay_wal_into( Arc::clone(&table_store), + &native_wal_reader(&table_store), &DbReaderOptions::default(), &core, &mut into_tables, @@ -2598,7 +2675,10 @@ mod tests { tokio::time::sleep(Duration::from_millis(20)).await; let result = reader.get(b"key").await.unwrap_err(); dbg!(&result); - assert_eq!(result.to_string(), "Unavailable error: io error (oops)"); + assert_eq!( + result.to_string(), + "Unavailable error: wal unavailable (io error)" + ); } #[tokio::test] @@ -2716,9 +2796,9 @@ mod tests { // Inject a failpoint on WAL probing before flushing so it is active // when the poller fires. With the buggy replay_new_wals=true, - // reestablish_checkpoint calls last_seen_wal_id() which probes WAL SSTs + // reestablish_checkpoint resolves the last WAL file by probing WAL SSTs // and hits this failpoint. With the fix (replay_new_wals=false), the - // WAL probing is skipped entirely. + // WAL probe is skipped entirely. fail_parallel::cfg( Arc::clone(&test_provider.fp_registry), "probe-wal-ssts", @@ -2739,7 +2819,7 @@ mod tests { // Wait for the manifest poller to see the changed L0 state and // reestablish the checkpoint. Without the fix, the poller crashes - // on the WAL listing failpoint. + // on the WAL probing failpoint. let timeout = Duration::from_secs(5); let start = tokio::time::Instant::now(); loop { @@ -2987,6 +3067,7 @@ mod tests { self.manifest_store(), self.table_store(), mode, + None, merge_operator, None, options, @@ -2998,6 +3079,10 @@ mod tests { } } + fn native_wal_reader(table_store: &Arc) -> crate::wal::reader::SlateDbWalReader { + crate::wal::reader::SlateDbWalReader::new(Arc::clone(table_store)) + } + fn immutable_memtable( recent_flushed_wal_id: u64, entries: Vec, @@ -3172,9 +3257,11 @@ mod tests { oracle.clone(), None, ); + let wal_reader = Arc::new(native_wal_reader(&table_store)); let inner = DbReaderInner { manifest_store, table_store, + wal_reader, options: DbReaderOptions { skip_wal_replay: true, ..DbReaderOptions::default() @@ -3258,9 +3345,11 @@ mod tests { oracle.clone(), None, ); + let wal_reader = Arc::new(native_wal_reader(&table_store)); DbReaderInner { manifest_store, table_store, + wal_reader, options: DbReaderOptions::default(), mode: DbReaderMode::ManagedCheckpoint, state: parking_lot::RwLock::new(Arc::new(prior_state)), diff --git a/slatedb/src/fence.rs b/slatedb/src/fence.rs index 399a982ae2..182d7ef752 100644 --- a/slatedb/src/fence.rs +++ b/slatedb/src/fence.rs @@ -20,6 +20,7 @@ pub(crate) struct WriterFencer { manifest_update_timeout: Duration, system_clock: Arc, task_executor: Arc, + wal_writer_init: Option>, #[cfg_attr(not(test), allow(dead_code))] fp_tx: FailPointTx, } @@ -38,6 +39,7 @@ impl WriterFencer { settings: &Settings, system_clock: Arc, task_executor: Arc, + wal_writer_init: Option>, ) -> Self { Self::new_with_fp_handle( closed_result_reader, @@ -46,6 +48,7 @@ impl WriterFencer { settings, system_clock, task_executor, + wal_writer_init, FailPointTx::dummy(), ) } @@ -57,6 +60,7 @@ impl WriterFencer { settings: &Settings, system_clock: Arc, task_executor: Arc, + wal_writer_init: Option>, fp_tx: FailPointTx, ) -> Self { Self { @@ -67,6 +71,7 @@ impl WriterFencer { manifest_update_timeout: settings.manifest_update_timeout, system_clock, task_executor, + wal_writer_init, fp_tx, } } @@ -76,24 +81,28 @@ impl WriterFencer { } /// Fences all writers with an older epoch than the provided `stored_manifest` by (1) writing - /// a new `FenceableManifest` with a bumped epoch, and (2) writing an empty WAL file that acts - /// as a barrier. Any parallel old writers will fail with `SlateDBError::Fenced` when trying - /// to "re-write" this file. Returns a `WriterFence` with the `FenceableManifest` and iterator - /// that must be replayed to recover up to the current epoch. + /// a new `FenceableManifest` with a bumped epoch, and (2) delegating WAL fencing and recovery + /// to the configured [`WriterInit`]. Returns the fenced manifest, the initialized WAL writer, + /// and the iterator that must be replayed to recover up to the current epoch. pub(crate) async fn fence( - self, + mut self, stored_manifest: StoredManifest, ) -> Result { - let wal_writer_init = WalWriterInit::load( - self.closed_result_reader.clone(), - self.recorder.clone(), - self.table_store.clone(), - self.wal_writer_init_options, - stored_manifest.manifest(), - self.task_executor.clone(), - self.fp_tx.clone(), - ) - .await?; + let wal_writer_init = match self.wal_writer_init.take() { + Some(wal_writer_init) => wal_writer_init, + None => Box::new( + WalWriterInit::load( + self.closed_result_reader.clone(), + self.recorder.clone(), + self.table_store.clone(), + self.wal_writer_init_options, + stored_manifest.manifest(), + self.task_executor.clone(), + self.fp_tx.clone(), + ) + .await?, + ), + }; let manifest = FenceableManifest::init_writer( stored_manifest, @@ -205,6 +214,7 @@ mod tests { &settings, system_clock.clone(), task_executor.clone(), + None, fp_tx, ); Self { @@ -283,6 +293,7 @@ mod tests { &MetricsRecorderHelper::noop(), Arc::new(DefaultSystemClock::new()), None, + None, ); gc.run_gc_once().await; // verify all regular (size > 0) wals up to wal_id are deleted (the wal diff --git a/slatedb/src/garbage_collector.rs b/slatedb/src/garbage_collector.rs index a4fa093242..ad9fe783ae 100644 --- a/slatedb/src/garbage_collector.rs +++ b/slatedb/src/garbage_collector.rs @@ -51,6 +51,7 @@ mod manifest_gc; pub mod stats; mod wal_gc; +use crate::wal::WalGc; pub(crate) use filter::retain_allowed_by_gc_filter; pub use filter::GcFilter; @@ -235,6 +236,7 @@ impl GarbageCollector { recorder: &MetricsRecorderHelper, system_clock: Arc, gc_filter: Option>, + wal_gc: Option>, ) -> Self { let stats = Arc::new(GcStats::new(recorder)); // The standalone GC lifecycle does not surface a closed result yet, so the @@ -245,30 +247,38 @@ impl GarbageCollector { system_clock.clone(), )); let wal_gc_task = options.wal_options.map(|wal_options| { - let wal_gc = Arc::new(SlateDbWalGc::new( - table_store.clone(), - stats.clone(), - wal_options, - WalGcMode::Regular, - gc_filter.clone(), - system_clock.clone(), - )); + let wal_gc = wal_gc.unwrap_or_else(|| { + Arc::new(SlateDbWalGc::new( + table_store.clone(), + stats.clone(), + WalGcMode::Regular, + gc_filter.clone(), + system_clock.clone(), + )) + }); WalGcTask::new( manifest_store.clone(), wal_gc, WalGcMode::Regular.resource(), + wal_options.min_age, + wal_options.dry_run, ) }); let wal_fence_gc_task = options.wal_fence_options.map(|wal_fence_options| { let wal_gc = Arc::new(SlateDbWalGc::new( table_store.clone(), stats.clone(), - wal_fence_options, WalGcMode::Fence, gc_filter.clone(), system_clock.clone(), )); - WalGcTask::new(manifest_store.clone(), wal_gc, WalGcMode::Fence.resource()) + WalGcTask::new( + manifest_store.clone(), + wal_gc, + WalGcMode::Fence.resource(), + wal_fence_options.min_age, + wal_fence_options.dry_run, + ) }); let compacted_gc_task = options.compacted_options.map(|compacted_options| { CompactedGcTask::new( @@ -1190,6 +1200,7 @@ mod tests { &MetricsRecorderHelper::noop(), Arc::new(DefaultSystemClock::default()), None, + None, ); gc.run_gc_once().await; @@ -1260,6 +1271,7 @@ mod tests { &helper, Arc::new(DefaultSystemClock::default()), None, + None, ); gc.run_gc_once().await; @@ -1325,6 +1337,7 @@ mod tests { &MetricsRecorderHelper::noop(), Arc::new(DefaultSystemClock::default()), None, + None, ); gc.run_gc_once().await; @@ -1405,6 +1418,7 @@ mod tests { &MetricsRecorderHelper::noop(), Arc::new(DefaultSystemClock::default()), None, + None, ); gc.run_gc_once().await; @@ -1869,6 +1883,7 @@ mod tests { recorder, Arc::new(DefaultSystemClock::default()), None, + None, ); gc.run_gc_once().await; @@ -1946,6 +1961,7 @@ mod tests { &recorder, Arc::new(DefaultSystemClock::default()), None, + None, ); // Send a WAL GC message. Correct behavior: only WAL GC runs. @@ -2018,6 +2034,7 @@ mod tests { &recorder, Arc::new(DefaultSystemClock::default()), None, + None, ); gc.run_gc_once().await; @@ -2069,6 +2086,7 @@ mod tests { &recorder, Arc::new(DefaultSystemClock::default()), None, + None, ); let intervals: Vec<_> = gc.tickers().into_iter().map(|def| def.interval).collect(); @@ -2124,6 +2142,7 @@ mod tests { &recorder, Arc::new(DefaultSystemClock::default()), None, + None, ); gc.start().expect("failed to start garbage collector"); gc.stop().await.expect("failed to stop garbage collector"); @@ -2457,6 +2476,7 @@ mod tests { Some(Arc::new(LocationGcFilter { allowed_locations: HashSet::new(), })), + None, ); // Run every directory GC task with candidates present for each task type. @@ -2558,6 +2578,7 @@ mod tests { Some(Arc::new(LocationGcFilter { allowed_locations: HashSet::from([path_resolver.sst_path(&allowed_wal_id)]), })), + None, ); // Run WAL GC with a filter that permits only one of the eligible WALs. @@ -2690,6 +2711,7 @@ mod tests { &recorder, Arc::new(DefaultSystemClock::default()), None, + None, ); gc.run_gc_once().await; diff --git a/slatedb/src/garbage_collector/wal_gc.rs b/slatedb/src/garbage_collector/wal_gc.rs index 3b455ed25c..ad6d2b5181 100644 --- a/slatedb/src/garbage_collector/wal_gc.rs +++ b/slatedb/src/garbage_collector/wal_gc.rs @@ -1,3 +1,4 @@ +use super::GcTask; use crate::manifest::Manifest; use crate::{ error::SlateDBError, @@ -8,14 +9,15 @@ use chrono::{DateTime, Utc}; use std::collections::BTreeMap; use std::ops::Bound; use std::sync::Arc; - -use super::GcTask; +use std::time::Duration; #[derive(Clone)] pub(crate) struct WalGcTask { manifest_store: Arc, wal_gc: Arc, resource: &'static str, + min_age: Duration, + dry_run: bool, } impl std::fmt::Debug for WalGcTask { @@ -31,11 +33,15 @@ impl WalGcTask { manifest_store: Arc, wal_gc: Arc, resource: &'static str, + min_age: Duration, + dry_run: bool, ) -> Self { Self { manifest_store, wal_gc, resource, + min_age, + dry_run, } } @@ -77,7 +83,7 @@ impl GcTask for WalGcTask { let referenced_ranges = Self::referenced_wal_ranges(latest_manifest.id, &active_manifests); self.wal_gc - .collect(referenced_ranges) + .collect(referenced_ranges, self.min_age, self.dry_run) .await .map_err(Into::into) } @@ -100,6 +106,7 @@ mod tests { use object_store::ObjectStore; use slatedb_common::clock::DefaultSystemClock; use std::sync::Mutex; + use std::time::Duration; use uuid::Uuid; #[derive(Default)] @@ -115,7 +122,12 @@ mod tests { #[async_trait] impl WalGc for RecordingWalGc { - async fn collect(&self, referenced_ranges: Vec) -> Result<(), WalError> { + async fn collect( + &self, + referenced_ranges: Vec, + _min_age: Duration, + _dry_run: bool, + ) -> Result<(), WalError> { self.calls.lock().unwrap().push(referenced_ranges); Ok(()) } @@ -178,7 +190,7 @@ mod tests { stored_manifest.update(dirty).await.unwrap(); let wal_gc = Arc::new(RecordingWalGc::default()); - let task = WalGcTask::new(manifest_store, wal_gc.clone(), "WAL"); + let task = WalGcTask::new(manifest_store, wal_gc.clone(), "WAL", Duration::ZERO, false); task.collect(Utc::now()).await.unwrap(); diff --git a/slatedb/src/wal/admin.rs b/slatedb/src/wal/admin.rs index 7d7546a6cb..855fa634b6 100644 --- a/slatedb/src/wal/admin.rs +++ b/slatedb/src/wal/admin.rs @@ -1,5 +1,4 @@ use crate::block_cache_policy::BlockCachePolicy; -use crate::config::GarbageCollectorDirectoryOptions; use crate::db_state::SsTableId; use crate::format::sst::SsTableFormat; use crate::garbage_collector::stats::GcStats; @@ -69,7 +68,7 @@ impl SlateDbWalAdmin { #[async_trait] impl WalAdmin for SlateDbWalAdmin { - fn garbage_collector(&self, path: &Path) -> Box { + fn garbage_collector(&self, path: &Path) -> Arc { let table_store = Arc::new(TableStore::new( ObjectStores::new(self.object_store.clone(), None), SsTableFormat::default(), @@ -78,26 +77,28 @@ impl WalAdmin for SlateDbWalAdmin { TableStoreKind::GC, BlockCachePolicy::default(), )); - Box::new(SlateDbWalGc::new( + Arc::new(SlateDbWalGc::new( table_store, Arc::new(GcStats::new(&MetricsRecorderHelper::noop())), - GarbageCollectorDirectoryOptions::default(), WalGcMode::Regular, None, Arc::new(DefaultSystemClock::new()), )) } - async fn delete_wal(&self, path: &Path) -> Result<(), WalError> { + async fn delete_wal(&self, path: &Path, dry_run: bool) -> Result, WalError> { // Collect the paths first so listing is complete before objects are removed. let wal_path = PathResolver::from_root(path.clone()).wal_path(); - for object_path in self.paths_under(&wal_path).await? { - self.object_store - .delete(&object_path) - .await - .map_err(|err| WalError::Unavailable(Arc::new(err)))?; + let paths = self.paths_under(&wal_path).await?; + if !dry_run { + for object_path in &paths { + self.object_store + .delete(object_path) + .await + .map_err(|err| WalError::Unavailable(Arc::new(err)))?; + } } - Ok(()) + Ok(paths.iter().map(Path::to_string).collect()) } async fn is_empty( @@ -190,7 +191,7 @@ mod tests { .await .unwrap(); - wal_admin.delete_wal(&db_path).await.unwrap(); + wal_admin.delete_wal(&db_path, false).await.unwrap(); assert!(matches!( object_store.head(&wal_object).await, diff --git a/slatedb/src/wal/gc.rs b/slatedb/src/wal/gc.rs index 55e4843ce4..baae19e943 100644 --- a/slatedb/src/wal/gc.rs +++ b/slatedb/src/wal/gc.rs @@ -1,4 +1,3 @@ -use crate::config::GarbageCollectorDirectoryOptions; use crate::db_state::SsTableId; use crate::garbage_collector::stats::GcStats; use crate::garbage_collector::{retain_allowed_by_gc_filter, GcFilter, GC_DELETE_CONCURRENCY}; @@ -12,6 +11,7 @@ use slatedb_common::clock::SystemClock; use slatedb_common::object_metadata::IdentifiedObjectMetadata; use std::ops::Bound; use std::sync::Arc; +use std::time::Duration; /// Selects which class of SlateDB WAL object is collected. /// @@ -42,7 +42,6 @@ impl WalGcMode { pub(crate) struct SlateDbWalGc { table_store: Arc, stats: Arc, - wal_options: GarbageCollectorDirectoryOptions, mode: WalGcMode, gc_filter: Option>, system_clock: Arc, @@ -51,7 +50,6 @@ pub(crate) struct SlateDbWalGc { impl std::fmt::Debug for SlateDbWalGc { fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { f.debug_struct("SlateDbWalGc") - .field("wal_options", &self.wal_options) .field("mode", &self.mode) .finish() } @@ -61,7 +59,6 @@ impl SlateDbWalGc { pub(crate) fn new( table_store: Arc, stats: Arc, - wal_options: GarbageCollectorDirectoryOptions, mode: WalGcMode, gc_filter: Option>, system_clock: Arc, @@ -69,7 +66,6 @@ impl SlateDbWalGc { Self { table_store, stats, - wal_options, mode, gc_filter, system_clock, @@ -106,15 +102,15 @@ impl SlateDbWalGc { after_start && before_end } - fn wal_sst_min_age(&self) -> chrono::Duration { - chrono::Duration::from_std(self.wal_options.min_age).expect("invalid duration") + fn wal_sst_min_age(&self, min_age: Duration) -> chrono::Duration { + chrono::Duration::from_std(min_age).expect("invalid duration") } /// Deletes the given WAL SSTs from the table store. /// /// In case of dryrun, the actual deletion doesn't happen. - async fn maybe_delete_wal_ssts(&self, sst_ids: Vec) { - if self.wal_options.dry_run { + async fn maybe_delete_wal_ssts(&self, sst_ids: Vec, dry_run: bool) { + if dry_run { if !sst_ids.is_empty() { log::info!( "dry run: skipping {} deletion [count={}]", @@ -156,9 +152,14 @@ impl SlateDbWalGc { #[async_trait] impl WalGc for SlateDbWalGc { - async fn collect(&self, referenced_ranges: Vec) -> Result<(), WalError> { + async fn collect( + &self, + referenced_ranges: Vec, + min_age: Duration, + dry_run: bool, + ) -> Result<(), WalError> { let utc_now = self.system_clock.now(); - let min_age = self.wal_sst_min_age(); + let min_age = self.wal_sst_min_age(min_age); let ssts_to_delete = self .table_store .list_wal_ssts(..) @@ -185,7 +186,7 @@ impl WalGc for SlateDbWalGc { .map(|wal_sst| wal_sst.id) .collect::>(); - self.maybe_delete_wal_ssts(sst_ids_to_delete).await; + self.maybe_delete_wal_ssts(sst_ids_to_delete, dry_run).await; Ok(()) } @@ -222,16 +223,10 @@ mod tests { table_store: Arc, clock: Arc, mode: WalGcMode, - min_age: Duration, ) -> SlateDbWalGc { SlateDbWalGc::new( table_store, Arc::new(GcStats::new(&MetricsRecorderHelper::noop())), - GarbageCollectorDirectoryOptions { - interval: None, - min_age, - dry_run: false, - }, mode, None, clock, @@ -297,14 +292,12 @@ mod tests { write_regular_wal(&table_store, wal_id).await; } make_all_wals_older_than(&table_store, &clock, Duration::ZERO).await; - let collector = build_collector( - table_store.clone(), - clock, - WalGcMode::Regular, - Duration::ZERO, - ); + let collector = build_collector(table_store.clone(), clock, WalGcMode::Regular); - collector.collect(protect_outer_wals()).await.unwrap(); + collector + .collect(protect_outer_wals(), Duration::ZERO, false) + .await + .unwrap(); assert_eq!(wal_ids(&table_store).await, vec![1, 4]); } @@ -316,14 +309,12 @@ mod tests { write_regular_wal(&table_store, 1).await; write_fence_wal(&table_store, 2).await; make_all_wals_older_than(&table_store, &clock, Duration::ZERO).await; - let collector = build_collector( - table_store.clone(), - clock, - WalGcMode::Regular, - Duration::ZERO, - ); + let collector = build_collector(table_store.clone(), clock, WalGcMode::Regular); - collector.collect(vec![]).await.unwrap(); + collector + .collect(vec![], Duration::ZERO, false) + .await + .unwrap(); assert_eq!(wal_ids(&table_store).await, vec![2]); } @@ -339,19 +330,14 @@ mod tests { .unwrap() .last_modified; let min_age = Duration::from_secs(60 * 60); - let collector = build_collector( - table_store.clone(), - clock.clone(), - WalGcMode::Regular, - min_age, - ); + let collector = build_collector(table_store.clone(), clock.clone(), WalGcMode::Regular); clock.set((last_modified + chrono::Duration::minutes(30)).timestamp_millis()); - collector.collect(vec![]).await.unwrap(); + collector.collect(vec![], min_age, false).await.unwrap(); assert_eq!(wal_ids(&table_store).await, vec![1]); clock.set((last_modified + chrono::Duration::minutes(61)).timestamp_millis()); - collector.collect(vec![]).await.unwrap(); + collector.collect(vec![], min_age, false).await.unwrap(); assert!(wal_ids(&table_store).await.is_empty()); } @@ -363,10 +349,12 @@ mod tests { write_fence_wal(&table_store, wal_id).await; } make_all_wals_older_than(&table_store, &clock, Duration::ZERO).await; - let collector = - build_collector(table_store.clone(), clock, WalGcMode::Fence, Duration::ZERO); + let collector = build_collector(table_store.clone(), clock, WalGcMode::Fence); - collector.collect(protect_outer_wals()).await.unwrap(); + collector + .collect(protect_outer_wals(), Duration::ZERO, false) + .await + .unwrap(); assert_eq!(wal_ids(&table_store).await, vec![1, 4]); } @@ -378,10 +366,12 @@ mod tests { write_fence_wal(&table_store, 1).await; write_regular_wal(&table_store, 2).await; make_all_wals_older_than(&table_store, &clock, Duration::ZERO).await; - let collector = - build_collector(table_store.clone(), clock, WalGcMode::Fence, Duration::ZERO); + let collector = build_collector(table_store.clone(), clock, WalGcMode::Fence); - collector.collect(vec![]).await.unwrap(); + collector + .collect(vec![], Duration::ZERO, false) + .await + .unwrap(); assert_eq!(wal_ids(&table_store).await, vec![2]); } @@ -397,19 +387,14 @@ mod tests { .unwrap() .last_modified; let min_age = Duration::from_secs(60 * 60); - let collector = build_collector( - table_store.clone(), - clock.clone(), - WalGcMode::Fence, - min_age, - ); + let collector = build_collector(table_store.clone(), clock.clone(), WalGcMode::Fence); clock.set((last_modified + chrono::Duration::minutes(30)).timestamp_millis()); - collector.collect(vec![]).await.unwrap(); + collector.collect(vec![], min_age, false).await.unwrap(); assert_eq!(wal_ids(&table_store).await, vec![1]); clock.set((last_modified + chrono::Duration::minutes(61)).timestamp_millis()); - collector.collect(vec![]).await.unwrap(); + collector.collect(vec![], min_age, false).await.unwrap(); assert!(wal_ids(&table_store).await.is_empty()); } } diff --git a/slatedb/src/wal/mod.rs b/slatedb/src/wal/mod.rs index 7b3828b935..fc11c2914b 100644 --- a/slatedb/src/wal/mod.rs +++ b/slatedb/src/wal/mod.rs @@ -8,9 +8,11 @@ use std::error::Error; use std::fmt::{Display, Formatter}; use std::ops::{Bound, Range}; use std::sync::Arc; +use std::time::Duration; pub(crate) mod admin; pub(crate) mod gc; +pub(crate) mod reader; #[cfg(test)] pub(crate) mod test_utils; pub(crate) mod wal_disabled; @@ -146,7 +148,7 @@ pub struct WriterInitResult { /// rows in WAL files between [`WriterManifest::replay_after_wal_id`] (exclusive) and the /// current end of the WAL. #[async_trait] -pub trait WriterInit { +pub trait WriterInit: Send + Sync + 'static { /// Fences the WAL and returns a [`WriterInitResult`] with a [`WalWriter`] and /// [`WalReplayIterator`] used to recover writes that have not yet been flushed to the tree. async fn fence_and_init( @@ -271,7 +273,7 @@ pub trait WalIterator: Send + 'static { /// API for reading from the WAL. Used by the Reader/ #[async_trait] -pub trait WalReader { +pub trait WalReader: Send + Sync + 'static { /// Returns an iterator over the specified range of WAL File IDs. The start of the range must /// not be `Unbounded`. If the end of the range is `Unbounded` then the returned iterator /// continues returning writes as new writes are appended to the WAL. Otherwise, it returns @@ -280,6 +282,11 @@ pub trait WalReader { &self, wal_file_id_range: WalFileRange, ) -> Result, WalError>; + + /// Returns the ID of the last WAL file currently present after `replay_after_wal_id`, or + /// `replay_after_wal_id` if no later WAL file is present. Implementations may use + /// `replay_after_wal_id` as a known lower bound when locating the end of the WAL. + async fn last_wal_file_id(&self, replay_after_wal_id: u64) -> Result; } /// Trait that defines the contract between SlateDB's garbage collector and a custom WAL @@ -291,7 +298,12 @@ pub trait WalGc: Send + Sync + 'static { /// Hook for garbage collecting the WAL. Takes a list of ranges of WAL Files that are currently /// referenced by some active Manifest. The implementation may delete any WAL File that is not /// included in the ranges in this list. - async fn collect(&self, referenced_ranges: Vec) -> Result<(), WalError>; + async fn collect( + &self, + referenced_ranges: Vec, + min_age: Duration, + dry_run: bool, + ) -> Result<(), WalError>; } /// Administrative operations for a WAL implementation. @@ -304,16 +316,20 @@ pub trait WalAdmin: Send + Sync + 'static { /// /// ## Returns /// A garbage collector that can remove unreferenced WAL files at `path`. - fn garbage_collector(&self, path: &Path) -> Box; + fn garbage_collector(&self, path: &Path) -> Arc; /// Deletes the WAL at `path`. /// /// ## Arguments /// - `path`: The database path whose WAL should be deleted. + /// - `dry_run`: If set to true, the implementation should just return the list of resources + /// that would be deleted without actually deleting anything. /// /// ## Returns - /// `Ok(())` after the WAL has been deleted, or a [`WalError`] if deletion fails. - async fn delete_wal(&self, path: &Path) -> Result<(), WalError>; + /// `Ok(resources)` after the WAL has been deleted, or a [`WalError`] if deletion fails, where + /// `resources` is a list of descriptions of resources that were deleted by this fn. This + /// list is used to display the output of deleting the WAL (e.g. in logs or tool output) + async fn delete_wal(&self, path: &Path, dry_run: bool) -> Result, WalError>; /// Given a path and WAL ID range, returns true if the WAL at that path is empty within the /// specified range. A WAL is empty if it holds no records. diff --git a/slatedb/src/wal/reader.rs b/slatedb/src/wal/reader.rs new file mode 100644 index 0000000000..be0dd54460 --- /dev/null +++ b/slatedb/src/wal/reader.rs @@ -0,0 +1,114 @@ +use std::sync::Arc; + +use async_trait::async_trait; + +use crate::iter::IterationOrder; +use crate::sst_iter::SstIteratorOptions; +use crate::tablestore::TableStore; +use crate::wal::{WalError, WalFileRange, WalIterator, WalReader}; +use crate::wal_replay::{WalIterator as WalReplayIterator, WalIteratorOptions}; + +pub(crate) struct SlateDbWalReader { + table_store: Arc, +} + +impl SlateDbWalReader { + pub(crate) fn new(table_store: Arc) -> Self { + Self { table_store } + } +} + +#[async_trait] +impl WalReader for SlateDbWalReader { + async fn iterator( + &self, + wal_file_id_range: WalFileRange, + ) -> Result, WalError> { + let wal_id_range = wal_file_id_range.try_into().map_err(|()| { + WalError::InternalError(Arc::new(std::io::Error::new( + std::io::ErrorKind::InvalidInput, + "native WAL reader requires an included start and excluded end", + ))) + })?; + let iterator = WalReplayIterator::range( + wal_id_range, + WalIteratorOptions { + sst_batch_size: 4, + sst_iter_options: SstIteratorOptions { + max_fetch_tasks: 1, + blocks_to_fetch: 256, + cache_blocks: true, + cache_metadata: false, + eager_spawn: true, + order: IterationOrder::Ascending, + prefix: None, + filter_context: None, + }, + }, + Arc::clone(&self.table_store), + )?; + Ok(Box::new(iterator)) + } + + async fn last_wal_file_id(&self, replay_after_wal_id: u64) -> Result { + Ok(self + .table_store + .last_seen_wal_id(replay_after_wal_id) + .await?) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::block_cache_policy::BlockCachePolicy; + use crate::config::{FlushOptions, FlushType}; + use crate::format::sst::SsTableFormat; + use crate::object_stores::ObjectStores; + use crate::tablestore::TableStoreKind; + use crate::types::ValueDeletable; + use crate::Db; + use object_store::memory::InMemory; + use object_store::{path::Path, ObjectStore}; + + #[tokio::test] + async fn test_native_wal_reader_trait() { + let object_store: Arc = Arc::new(InMemory::new()); + let path = Path::from("/test_native_wal_reader_trait"); + let db = Db::open(path.clone(), Arc::clone(&object_store)) + .await + .unwrap(); + db.put(b"key", b"value").await.unwrap(); + db.flush_with_options(FlushOptions { + flush_type: FlushType::Wal, + }) + .await + .unwrap(); + + let table_store = Arc::new(TableStore::new( + ObjectStores::new(object_store, None), + SsTableFormat::default(), + path, + None, + TableStoreKind::Reader, + BlockCachePolicy::default(), + )); + let wal_reader = SlateDbWalReader::new(table_store); + let last_wal_id = wal_reader.last_wal_file_id(0).await.unwrap(); + let mut iterator = wal_reader + .iterator((1..last_wal_id + 1).into()) + .await + .unwrap(); + + let mut rows = Vec::new(); + while let Some(wal_rows) = iterator.next().await.unwrap() { + rows.extend(wal_rows.rows); + } + assert_eq!(rows.len(), 1); + assert_eq!(rows[0].key.as_ref(), b"key"); + assert!(matches!( + &rows[0].value, + ValueDeletable::Value(value) if value.as_ref() == b"value" + )); + } +} diff --git a/slatedb/src/wal_replay.rs b/slatedb/src/wal_replay.rs index 1fe70eaafc..1d09773f29 100644 --- a/slatedb/src/wal_replay.rs +++ b/slatedb/src/wal_replay.rs @@ -77,6 +77,7 @@ pub(crate) struct WalReplayIterator { } impl WalReplayIterator { + #[cfg(test)] pub(crate) fn range( wal_id_range: Range, db_state: &ManifestCore, diff --git a/slatedb/tests/custom_wal.rs b/slatedb/tests/custom_wal.rs new file mode 100644 index 0000000000..14239a9773 --- /dev/null +++ b/slatedb/tests/custom_wal.rs @@ -0,0 +1,524 @@ +use std::collections::{BTreeMap, VecDeque}; +use std::ops::Bound; +use std::sync::Arc; +use std::time::Duration; + +use async_trait::async_trait; +use parking_lot::Mutex; +use slatedb::admin::{Admin, CloneSourceSpec}; +use slatedb::config::{ + CloseOptions, DbReaderOptions, FlushOptions, FlushType, GarbageCollectorDirectoryOptions, + GarbageCollectorOptions, Settings, +}; +use slatedb::object_store::memory::InMemory; +use slatedb::object_store::path::Path; +use slatedb::object_store::ObjectStore; +use slatedb::wal::{ + FlushResultFuture, WalAdmin, WalError, WalEvent, WalFileRange, WalGc, WalIterator, WalObserver, + WalReader, WalRows, WalStatus, WalStatusListener, WalWriter, WriterInit, WriterInitResult, + WriterManifest, +}; +use slatedb::{Db, DbReader, DbReaderMode, GarbageCollectorBuilder, RowEntry, VersionedManifest}; + +/// A deliberately small WAL implementation used to exercise the public pluggable-WAL API. +/// Every call to `WalWriter::append` inserts one write batch under a new WAL file ID. +#[derive(Clone, Default)] +struct BTreeMapWal { + files: Arc>>>, +} + +impl BTreeMapWal { + fn file_ids(&self) -> Vec { + self.files.lock().keys().copied().collect() + } + + fn snapshot(&self) -> BTreeMap> { + self.files.lock().clone() + } + + fn open_status(&self) -> WalStatus { + let files = self.files.lock(); + let last_flushed_wal_id = files.last_key_value().map(|(id, _)| *id).unwrap_or(0); + let last_flushed_seq = files + .last_key_value() + .and_then(|(_, batch)| batch.iter().map(|row| row.seq).max()); + WalStatus { + closed_reason: None, + estimated_bytes: 0, + last_flushed_wal_id, + last_flushed_seq, + buffered_wal_entries_count: 0, + } + } +} + +struct BTreeMapWalIterator { + batches: VecDeque<(u64, Vec)>, +} + +#[async_trait] +impl WalIterator for BTreeMapWalIterator { + async fn next(&mut self) -> Result, WalError> { + Ok(self.batches.pop_front().map(|(wal_file_id, rows)| WalRows { + rows, + last_consumed_wal_file_id: wal_file_id, + })) + } +} + +#[async_trait] +impl WalReader for BTreeMapWal { + async fn iterator( + &self, + wal_file_id_range: WalFileRange, + ) -> Result, WalError> { + let WalFileRange(start, end) = wal_file_id_range; + let batches = self + .files + .lock() + .range((start, end)) + .map(|(id, batch)| (*id, batch.clone())) + .collect(); + Ok(Box::new(BTreeMapWalIterator { batches })) + } + + async fn last_wal_file_id(&self, replay_after_wal_id: u64) -> Result { + Ok(self + .files + .lock() + .range((Bound::Excluded(replay_after_wal_id), Bound::Unbounded)) + .next_back() + .map(|(id, _)| *id) + .unwrap_or(replay_after_wal_id)) + } +} + +struct ObserverState { + status: WalStatus, + listeners: Vec, +} + +#[derive(Clone)] +struct BTreeMapWalObserver { + state: Arc>, +} + +impl BTreeMapWalObserver { + fn status(&self) -> Result { + let status = self.state.lock().status.clone(); + if status.closed_reason.is_some() { + Err(status) + } else { + Ok(status) + } + } +} + +impl WalObserver for BTreeMapWalObserver { + fn status(&self) -> Result { + self.status() + } + + fn subscribe(&self, listener: WalStatusListener) -> Result<(), WalError> { + self.state.lock().listeners.push(listener); + Ok(()) + } +} + +struct BTreeMapWalWriter { + wal: BTreeMapWal, + observer: BTreeMapWalObserver, +} + +impl BTreeMapWalWriter { + fn new(wal: BTreeMapWal) -> Self { + let observer = BTreeMapWalObserver { + state: Arc::new(Mutex::new(ObserverState { + status: wal.open_status(), + listeners: Vec::new(), + })), + }; + Self { wal, observer } + } + + fn publish(&self, event: WalEvent) { + let listeners = self.observer.state.lock().listeners.clone(); + for listener in listeners { + listener(event.clone()); + } + } +} + +#[async_trait] +impl WalWriter for BTreeMapWalWriter { + async fn append(&mut self, write_batch: &[RowEntry]) -> Result<(), WalError> { + if self.observer.state.lock().status.closed_reason.is_some() { + return Err(WalError::Closed); + } + + let wal_file_id = { + let mut files = self.wal.files.lock(); + let wal_file_id = match files.last_key_value() { + Some((last_id, _)) => last_id.checked_add(1).ok_or_else(|| { + WalError::InternalError(Arc::new(std::io::Error::other("WAL file ID overflow"))) + })?, + None => 1, + }; + files.insert(wal_file_id, write_batch.to_vec()); + wal_file_id + }; + + let status = { + let mut state = self.observer.state.lock(); + state.status.last_flushed_wal_id = wal_file_id; + state.status.last_flushed_seq = write_batch.iter().map(|row| row.seq).max(); + state.status.clone() + }; + self.publish(WalEvent::WalFlushed(status)); + Ok(()) + } + + async fn flush(&mut self) -> Result { + Ok(Box::pin(async { Ok(()) })) + } + + fn observer(&self) -> Box { + Box::new(self.observer.clone()) + } + + fn status(&self) -> Result { + self.observer.status() + } + + async fn close(&mut self) -> Result<(), WalError> { + let status = { + let mut state = self.observer.state.lock(); + if state.status.closed_reason.is_some() { + return Ok(()); + } + state.status.closed_reason = Some(WalError::Closed); + state.status.clone() + }; + self.publish(WalEvent::WalClosed(status)); + Ok(()) + } +} + +#[async_trait] +impl WriterInit for BTreeMapWal { + async fn fence_and_init( + &self, + manifest: &mut WriterManifest, + ) -> Result { + let replay_after_wal_id = manifest.replay_after_wal_id(); + let start = replay_after_wal_id.checked_add(1).ok_or_else(|| { + WalError::InternalError(Arc::new(std::io::Error::other("WAL replay range overflow"))) + })?; + let end = self + .last_wal_file_id(replay_after_wal_id) + .await? + .checked_add(1) + .ok_or_else(|| { + WalError::InternalError(Arc::new(std::io::Error::other( + "WAL replay range overflow", + ))) + })?; + let replay_iterator = self.iterator((start..end.max(start)).into()).await?; + + Ok(WriterInitResult { + replay_iterator, + wal_writer: Box::new(BTreeMapWalWriter::new(self.clone())), + }) + } +} + +fn range_contains(range: &WalFileRange, wal_file_id: u64) -> bool { + let starts_before = match range.0 { + Bound::Included(start) => wal_file_id >= start, + Bound::Excluded(start) => wal_file_id > start, + Bound::Unbounded => true, + }; + let ends_after = match range.1 { + Bound::Included(end) => wal_file_id <= end, + Bound::Excluded(end) => wal_file_id < end, + Bound::Unbounded => true, + }; + starts_before && ends_after +} + +#[async_trait] +impl WalGc for BTreeMapWal { + async fn collect( + &self, + referenced_ranges: Vec, + _min_age: Duration, + dry_run: bool, + ) -> Result<(), WalError> { + if !dry_run { + self.files.lock().retain(|wal_file_id, _| { + referenced_ranges + .iter() + .any(|range| range_contains(range, *wal_file_id)) + }); + } + Ok(()) + } +} + +/// Supplies a separate `BTreeMapWal` for each database path so clone administration can copy +/// the source WAL into the clone's WAL namespace. +#[derive(Clone, Default)] +struct BTreeMapWalAdmin { + wals: Arc>>, +} + +impl BTreeMapWalAdmin { + fn wal(&self, path: &Path) -> BTreeMapWal { + self.wals + .lock() + .entry(path.to_string()) + .or_default() + .clone() + } +} + +#[async_trait] +impl WalAdmin for BTreeMapWalAdmin { + fn garbage_collector(&self, path: &Path) -> Arc { + Arc::new(self.wal(path)) + } + + async fn delete_wal(&self, path: &Path, dry_run: bool) -> Result, WalError> { + let key = path.to_string(); + let exists = self.wals.lock().contains_key(&key); + if exists && !dry_run { + self.wals.lock().remove(&key); + } + Ok(exists + .then(|| format!("btree-map-wal:{key}")) + .into_iter() + .collect()) + } + + async fn is_empty( + &self, + path: &Path, + _replay_after_wal_id: u64, + _wal_id_last_seen: u64, + ) -> Result { + Ok(self.wal(path).files.lock().values().all(Vec::is_empty)) + } + + async fn clone_wal( + &self, + from_path: &Path, + from_manifest: VersionedManifest, + to_path: &Path, + ) -> Result<(u64, u64), WalError> { + let replay_after_wal_id = from_manifest.replay_after_wal_id(); + let copied = self + .wal(from_path) + .files + .lock() + .range((Bound::Excluded(replay_after_wal_id), Bound::Unbounded)) + .map(|(id, batch)| (*id, batch.clone())) + .collect::>(); + let last_wal_file_id = copied + .last_key_value() + .map(|(id, _)| *id) + .unwrap_or(replay_after_wal_id); + *self.wal(to_path).files.lock() = copied; + Ok((replay_after_wal_id, last_wal_file_id)) + } +} + +fn test_settings() -> Settings { + Settings { + flush_interval: None, + compactor_options: None, + garbage_collector_options: None, + ..Settings::default() + } +} + +async fn open_db(path: Path, object_store: Arc, wal: BTreeMapWal) -> Db { + Db::builder(path, object_store) + .with_settings(test_settings()) + .with_wal_writer(Box::new(wal)) + .build() + .await + .expect("failed to open database with custom WAL") +} + +async fn close_without_memtable_flush(db: &Db) { + db.close_with_options(CloseOptions::default().with_flush_type(None)) + .await + .expect("failed to close database") +} + +#[tokio::test] +async fn custom_wal_basic_write_and_recovery() { + let object_store: Arc = Arc::new(InMemory::new()); + let path = Path::from("/custom-wal/basic-recovery"); + let wal = BTreeMapWal::default(); + + let db = open_db(path.clone(), Arc::clone(&object_store), wal.clone()).await; + db.put(b"key-1", b"value-1").await.expect("put failed"); + db.put(b"key-2", b"value-2").await.expect("put failed"); + + assert_eq!(wal.file_ids(), vec![1, 2]); + assert!(wal.snapshot().values().all(|batch| batch.len() == 1)); + close_without_memtable_flush(&db).await; + + let recovered = open_db(path, object_store, wal).await; + assert_eq!( + recovered + .get(b"key-1") + .await + .expect("get failed") + .as_deref(), + Some(b"value-1".as_slice()) + ); + assert_eq!( + recovered + .get(b"key-2") + .await + .expect("get failed") + .as_deref(), + Some(b"value-2".as_slice()) + ); + close_without_memtable_flush(&recovered).await; +} + +#[tokio::test] +async fn custom_wal_db_reader_serves_wal_data() { + let object_store: Arc = Arc::new(InMemory::new()); + let path = Path::from("/custom-wal/db-reader"); + let wal = BTreeMapWal::default(); + let db = open_db(path.clone(), Arc::clone(&object_store), wal.clone()).await; + db.put(b"reader-key", b"reader-value") + .await + .expect("put failed"); + + let reader = DbReader::builder(path, object_store) + .with_reader_mode(DbReaderMode::FollowLatest) + .with_options(DbReaderOptions { + manifest_poll_interval: Duration::from_secs(60), + ..DbReaderOptions::default() + }) + .with_wal_reader(Arc::new(wal)) + .build() + .await + .expect("failed to open database reader"); + assert_eq!( + reader + .get(b"reader-key") + .await + .expect("reader get failed") + .as_deref(), + Some(b"reader-value".as_slice()) + ); + + reader.close().await.expect("failed to close reader"); + close_without_memtable_flush(&db).await; +} + +#[tokio::test] +async fn custom_wal_garbage_collects_unused_ranges() { + let object_store: Arc = Arc::new(InMemory::new()); + let path = Path::from("/custom-wal/gc"); + let wal = BTreeMapWal::default(); + let db = open_db(path.clone(), Arc::clone(&object_store), wal.clone()).await; + + for id in 1..=3 { + db.put( + format!("key-{id}").as_bytes(), + format!("value-{id}").as_bytes(), + ) + .await + .expect("put failed"); + } + db.flush_with_options(FlushOptions { + flush_type: FlushType::MemTable, + }) + .await + .expect("memtable flush failed"); + db.put(b"key-4", b"value-4").await.expect("put failed"); + close_without_memtable_flush(&db).await; + assert_eq!(wal.file_ids(), vec![1, 2, 3, 4]); + + let gc = GarbageCollectorBuilder::new(path, object_store) + .with_wal_gc(Arc::new(wal.clone())) + .with_options(GarbageCollectorOptions { + manifest_options: None, + wal_options: Some(GarbageCollectorDirectoryOptions { + interval: None, + min_age: Duration::ZERO, + dry_run: false, + }), + wal_fence_options: None, + compacted_options: None, + compactions_options: None, + detach_options: None, + ..GarbageCollectorOptions::default() + }) + .build(); + gc.run_gc_once().await; + + // The latest manifest references the replay boundary (3) and everything after it. + assert_eq!(wal.file_ids(), vec![3, 4]); +} + +#[tokio::test] +async fn custom_wal_clone_has_the_same_data() { + let object_store: Arc = Arc::new(InMemory::new()); + let source_path = Path::from("/custom-wal/clone-source"); + let clone_path = Path::from("/custom-wal/clone-destination"); + let wal_admin = BTreeMapWalAdmin::default(); + let source_wal = wal_admin.wal(&source_path); + + let source_db = open_db( + source_path.clone(), + Arc::clone(&object_store), + source_wal.clone(), + ) + .await; + source_db + .put(b"clone-key-1", b"clone-value-1") + .await + .expect("put failed"); + source_db + .put(b"clone-key-2", b"clone-value-2") + .await + .expect("put failed"); + close_without_memtable_flush(&source_db).await; + + Admin::builder(clone_path.clone(), Arc::clone(&object_store)) + .with_wal_admin(Arc::new(wal_admin.clone())) + .build() + .create_clone_builder_from_source(CloneSourceSpec::new(source_path)) + .build() + .await + .expect("failed to create clone"); + + let clone_wal = wal_admin.wal(&clone_path); + assert_eq!(clone_wal.snapshot(), source_wal.snapshot()); + let clone_db = open_db(clone_path, object_store, clone_wal).await; + assert_eq!( + clone_db + .get(b"clone-key-1") + .await + .expect("get failed") + .as_deref(), + Some(b"clone-value-1".as_slice()) + ); + assert_eq!( + clone_db + .get(b"clone-key-2") + .await + .expect("get failed") + .as_deref(), + Some(b"clone-value-2".as_slice()) + ); + close_without_memtable_flush(&clone_db).await; +} From 4b45c2f01fda568eff47e50012a63f9ddf1517a8 Mon Sep 17 00:00:00 2001 From: anchor Date: Thu, 13 Aug 2026 12:30:08 +0800 Subject: [PATCH 31/65] test(wal_replay): fix write_wal helper to respect max_entries (#2023) --- slatedb/src/wal_replay.rs | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/slatedb/src/wal_replay.rs b/slatedb/src/wal_replay.rs index 1d09773f29..e286e6f14d 100644 --- a/slatedb/src/wal_replay.rs +++ b/slatedb/src/wal_replay.rs @@ -1333,7 +1333,8 @@ mod tests { ) -> Result { let mut writer = table_store.table_writer(SsTableId::Wal(wal_id)); let mut next_seq = next_seq; - while next_seq < next_seq + (max_entries as u64) { + let end_seq = next_seq + (max_entries as u64); + while next_seq < end_seq { let Some((key, value)) = entries.next() else { break; }; From f88be86d17ac53260d3684edbc8f82811d945b5c Mon Sep 17 00:00:00 2001 From: Rui Fan <1996fanrui@gmail.com> Date: Thu, 13 Aug 2026 15:52:14 +0200 Subject: [PATCH 32/65] Enable unused_qualifications and drop redundant path prefixes (#2020) --- Cargo.toml | 1 + bindings/uniffi/src/admin.rs | 2 +- examples/src/create_snapshot.rs | 2 +- examples/src/scan_snapshot.rs | 2 +- slatedb-bencher/src/db.rs | 6 +- slatedb-bencher/src/transactions.rs | 2 +- slatedb-common/Cargo.toml | 3 + slatedb-common/src/clock.rs | 14 +-- slatedb-txn-obj/Cargo.toml | 3 + slatedb-txn-obj/src/object_store.rs | 4 +- slatedb/benches/db_transaction.rs | 2 +- slatedb/benches/write_batch.rs | 2 +- slatedb/src/batch.rs | 42 +++---- slatedb/src/batch_write.rs | 11 +- slatedb/src/bytes_generator.rs | 2 +- slatedb/src/bytes_range.rs | 103 ++++++++---------- .../src/cached_object_store/object_store.rs | 13 +-- slatedb/src/cached_object_store/storage_fs.rs | 20 ++-- slatedb/src/clone.rs | 4 +- slatedb/src/compaction_execute_bench.rs | 3 +- slatedb/src/compaction_worker.rs | 6 +- slatedb/src/compactions_store.rs | 5 +- slatedb/src/compactor.rs | 65 +++++------ slatedb/src/compactor_state.rs | 21 ++-- slatedb/src/config.rs | 2 +- slatedb/src/db.rs | 89 +++++---------- slatedb/src/db/builder.rs | 4 +- slatedb/src/db_cache/mod.rs | 8 +- slatedb/src/db_cache/serde.rs | 6 +- slatedb/src/db_iter.rs | 4 +- slatedb/src/db_reader.rs | 7 +- slatedb/src/db_state.rs | 77 ++++++------- slatedb/src/db_status.rs | 8 +- slatedb/src/dispatcher.rs | 2 +- slatedb/src/error.rs | 4 +- slatedb/src/filter.rs | 2 +- slatedb/src/flatbuffer_types.rs | 64 +++++------ slatedb/src/format/block.rs | 2 +- slatedb/src/format/row.rs | 8 +- slatedb/src/format/sst.rs | 12 +- slatedb/src/garbage_collector.rs | 60 ++++------ slatedb/src/lib.rs | 2 +- slatedb/src/manifest/mod.rs | 14 +-- slatedb/src/manifest/store.rs | 11 +- slatedb/src/mem_table.rs | 7 +- .../src/memtable_flusher/manifest_writer.rs | 4 +- slatedb/src/memtable_flusher/mod.rs | 20 ++-- slatedb/src/memtable_flusher/tracker.rs | 9 +- slatedb/src/memtable_flusher/uploader.rs | 17 +-- slatedb/src/merge_operator.rs | 6 +- slatedb/src/paths.rs | 4 +- slatedb/src/reader.rs | 12 +- slatedb/src/retention_iterator.rs | 6 +- slatedb/src/retrying_object_store.rs | 28 ++--- slatedb/src/size_tiered_compaction.rs | 2 +- slatedb/src/sst_builder.rs | 2 +- slatedb/src/sst_iter.rs | 2 +- slatedb/src/sst_reader.rs | 6 +- slatedb/src/sst_stats.rs | 2 +- slatedb/src/tablestore.rs | 8 +- slatedb/src/test_utils.rs | 18 +-- slatedb/src/types.rs | 20 ++-- slatedb/src/utils.rs | 21 ++-- slatedb/src/wal/wal_sst_builder.rs | 2 +- slatedb/src/wal_buffer.rs | 12 +- 65 files changed, 407 insertions(+), 525 deletions(-) diff --git a/Cargo.toml b/Cargo.toml index 7953fb6efb..b4a888816b 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -110,3 +110,4 @@ unexpected_cfgs = { level = "allow", check-cfg = [ 'cfg(tokio_unstable)', ] } unreachable_pub = "warn" +unused_qualifications = "warn" diff --git a/bindings/uniffi/src/admin.rs b/bindings/uniffi/src/admin.rs index ad5d48b9a6..ba0f795278 100644 --- a/bindings/uniffi/src/admin.rs +++ b/bindings/uniffi/src/admin.rs @@ -170,7 +170,7 @@ impl Admin { } /// Deletes the checkpoint with the specified id. - pub async fn delete_checkpoint(&self, id: String) -> Result<(), crate::Error> { + pub async fn delete_checkpoint(&self, id: String) -> Result<(), Error> { self.inner .delete_checkpoint(try_checkpoint_id_from_str(&id)?) .await diff --git a/examples/src/create_snapshot.rs b/examples/src/create_snapshot.rs index 982ea360d8..29c3b7c89d 100644 --- a/examples/src/create_snapshot.rs +++ b/examples/src/create_snapshot.rs @@ -5,7 +5,7 @@ use std::sync::Arc; #[tokio::main] async fn main() -> Result<(), Error> { // Initialize database - let object_store = Arc::new(slatedb::object_store::memory::InMemory::new()); + let object_store = Arc::new(object_store::memory::InMemory::new()); let db = Db::builder("my_db", object_store) .with_settings(Settings::default()) .build() diff --git a/examples/src/scan_snapshot.rs b/examples/src/scan_snapshot.rs index 9d4441d01e..706702697b 100644 --- a/examples/src/scan_snapshot.rs +++ b/examples/src/scan_snapshot.rs @@ -5,7 +5,7 @@ use std::sync::Arc; #[tokio::main] async fn main() -> Result<(), Error> { // Initialize database - let object_store = Arc::new(slatedb::object_store::memory::InMemory::new()); + let object_store = Arc::new(object_store::memory::InMemory::new()); let db = Db::builder("my_db", object_store) .with_settings(Settings::default()) .build() diff --git a/slatedb-bencher/src/db.rs b/slatedb-bencher/src/db.rs index bacda18249..46717a1bb0 100644 --- a/slatedb-bencher/src/db.rs +++ b/slatedb-bencher/src/db.rs @@ -87,7 +87,7 @@ impl RandomKeyGenerator { pub fn new(key_bytes: usize) -> Self { Self { key_len_bytes: key_bytes, - rng: rand_xorshift::XorShiftRng::from_os_rng(), + rng: XorShiftRng::from_os_rng(), used_keys: Vec::new(), } } @@ -134,7 +134,7 @@ impl FixedSetKeyGenerator { } Self { keys, - rng: rand_xorshift::XorShiftRng::from_os_rng(), + rng: XorShiftRng::from_os_rng(), used_keys: Vec::new(), } } @@ -275,7 +275,7 @@ impl Task { /// This method runs a loop, generating a key (and value if needed), and /// then either puts the key/value pair or gets the key. async fn run(&mut self) { - let mut random = rand_xorshift::XorShiftRng::from_os_rng(); + let mut random = XorShiftRng::from_os_rng(); let mut puts = 0u64; let mut puts_bytes = 0u64; let mut gets = 0u64; diff --git a/slatedb-bencher/src/transactions.rs b/slatedb-bencher/src/transactions.rs index 9b621ca565..d29c7d9a76 100644 --- a/slatedb-bencher/src/transactions.rs +++ b/slatedb-bencher/src/transactions.rs @@ -167,7 +167,7 @@ impl TransactionTask { /// /// This method runs a loop, executing transactions with multiple operations. async fn run(&mut self) { - let mut random = rand_xorshift::XorShiftRng::from_os_rng(); + let mut random = XorShiftRng::from_os_rng(); let mut commits = 0u64; let mut aborts = 0u64; let mut conflicts = 0u64; diff --git a/slatedb-common/Cargo.toml b/slatedb-common/Cargo.toml index 12d176e15b..0e5de9735c 100644 --- a/slatedb-common/Cargo.toml +++ b/slatedb-common/Cargo.toml @@ -23,3 +23,6 @@ test-util = ["tokio/test-util"] [dev-dependencies] tokio = { workspace = true, features = ["macros", "rt", "time", "test-util"] } + +[lints] +workspace = true diff --git a/slatedb-common/src/clock.rs b/slatedb-common/src/clock.rs index 2c38a6fe55..a5ae8babac 100644 --- a/slatedb-common/src/clock.rs +++ b/slatedb-common/src/clock.rs @@ -265,7 +265,7 @@ mod tests { #[tokio::test] #[cfg(feature = "test-util")] async fn test_mock_system_clock_set_now() { - let clock = std::sync::Arc::new(MockSystemClock::new()); + let clock = Arc::new(MockSystemClock::new()); // Test positive timestamp let positive_ts = 1625097600000i64; // 2021-07-01T00:00:00Z in milliseconds @@ -289,7 +289,7 @@ mod tests { #[tokio::test] #[cfg(feature = "test-util")] async fn test_mock_system_clock_advance() { - let clock = std::sync::Arc::new(MockSystemClock::new()); + let clock = Arc::new(MockSystemClock::new()); let initial_ts = 1000; // Set initial time @@ -310,7 +310,7 @@ mod tests { #[tokio::test] #[cfg(feature = "test-util")] async fn test_mock_system_clock_sleep() { - let clock = std::sync::Arc::new(MockSystemClock::new()); + let clock = Arc::new(MockSystemClock::new()); let initial_ts = 2000; // Set initial time @@ -352,7 +352,7 @@ mod tests { #[tokio::test] #[cfg(feature = "test-util")] async fn test_mock_system_clock_ticker() { - let clock = std::sync::Arc::new(MockSystemClock::new()); + let clock = Arc::new(MockSystemClock::new()); let tick_duration = Duration::from_millis(100); // Create a ticker @@ -388,7 +388,7 @@ mod tests { #[tokio::test(start_paused = true)] async fn test_default_system_clock_now() { - let clock = std::sync::Arc::new(DefaultSystemClock::new()); + let clock = Arc::new(DefaultSystemClock::new()); // Record initial time let initial_now = clock.now(); @@ -409,7 +409,7 @@ mod tests { #[tokio::test(start_paused = true)] #[cfg(feature = "test-util")] async fn test_default_system_clock_advance() { - let clock = std::sync::Arc::new(DefaultSystemClock::new()); + let clock = Arc::new(DefaultSystemClock::new()); let start = clock.now(); let duration = Duration::from_millis(500); clock.clone().advance(duration).await; @@ -424,7 +424,7 @@ mod tests { #[tokio::test(start_paused = true)] async fn test_default_system_clock_ticker() { - let clock = std::sync::Arc::new(DefaultSystemClock::new()); + let clock = Arc::new(DefaultSystemClock::new()); let tick_duration = Duration::from_millis(10); // Create a ticker diff --git a/slatedb-txn-obj/Cargo.toml b/slatedb-txn-obj/Cargo.toml index f15ec8b5b0..6675075d3e 100644 --- a/slatedb-txn-obj/Cargo.toml +++ b/slatedb-txn-obj/Cargo.toml @@ -24,3 +24,6 @@ test-util = [] [dev-dependencies] tempfile = { workspace = true } tokio = { workspace = true, features = ["macros", "rt", "time"] } + +[lints] +workspace = true diff --git a/slatedb-txn-obj/src/object_store.rs b/slatedb-txn-obj/src/object_store.rs index caf667621c..de0e7703ad 100644 --- a/slatedb-txn-obj/src/object_store.rs +++ b/slatedb-txn-obj/src/object_store.rs @@ -7,7 +7,7 @@ use bytes::Bytes; use futures::StreamExt; use log::{debug, error, warn}; use object_store::path::Path; -use object_store::Error::AlreadyExists; +use object_store::Error::{AlreadyExists, Precondition}; use object_store::{ Error, GetOptions, ObjectStore, ObjectStoreExt, PutMode, PutOptions, PutPayload, UpdateVersion, }; @@ -337,7 +337,7 @@ impl BoundaryObject for ObjectStoreBoundaryObject { return Ok(()); } // Try again if the boundary was concurrently updated by another process. - Err(Error::AlreadyExists { .. } | Error::Precondition { .. }) => { + Err(AlreadyExists { .. } | Precondition { .. }) => { // Refresh the cache so re-attempts always use the fresh boundary. self.read_boundary().await?; } diff --git a/slatedb/benches/db_transaction.rs b/slatedb/benches/db_transaction.rs index edb3f33f99..1bc7aab609 100644 --- a/slatedb/benches/db_transaction.rs +++ b/slatedb/benches/db_transaction.rs @@ -75,7 +75,7 @@ fn key(index: usize) -> Bytes { fn value(index: usize) -> Bytes { let mut value = vec![0; VALUE_SIZE]; - value[..std::mem::size_of::()].copy_from_slice(&index.to_le_bytes()); + value[..size_of::()].copy_from_slice(&index.to_le_bytes()); Bytes::from(value) } diff --git a/slatedb/benches/write_batch.rs b/slatedb/benches/write_batch.rs index 9bb206922d..b9bc7c85de 100644 --- a/slatedb/benches/write_batch.rs +++ b/slatedb/benches/write_batch.rs @@ -72,7 +72,7 @@ fn key(index: usize) -> Bytes { fn value(index: usize) -> Bytes { let mut value = vec![0; VALUE_SIZE]; - value[..std::mem::size_of::()].copy_from_slice(&index.to_le_bytes()); + value[..size_of::()].copy_from_slice(&index.to_le_bytes()); Bytes::from(value) } diff --git a/slatedb/src/batch.rs b/slatedb/src/batch.rs index 9a7244635f..836e32b7c4 100644 --- a/slatedb/src/batch.rs +++ b/slatedb/src/batch.rs @@ -455,15 +455,15 @@ impl WriteBatchIterator { #[async_trait] impl RowEntryIterator for WriteBatchIterator { - async fn init(&mut self) -> Result<(), crate::error::SlateDBError> { + async fn init(&mut self) -> Result<(), SlateDBError> { Ok(()) } - async fn next(&mut self) -> Result, crate::error::SlateDBError> { + async fn next(&mut self) -> Result, SlateDBError> { Ok(self.iter.next()) } - async fn seek(&mut self, next_key: &[u8]) -> Result<(), crate::error::SlateDBError> { + async fn seek(&mut self, next_key: &[u8]) -> Result<(), SlateDBError> { while let Some(entry) = self.iter.peek() { if match self.ordering { IterationOrder::Ascending => entry.key.as_ref() < next_key, @@ -1326,8 +1326,8 @@ mod tests { ]; assert_iterator(&mut iter, expected).await; - let merge_operator = Some(std::sync::Arc::new(StringConcatMergeOperator) - as crate::merge_operator::MergeOperatorType); + let merge_operator = + Some(std::sync::Arc::new(StringConcatMergeOperator) as MergeOperatorType); let (result, _, _) = batch .extract_entries(100, 1000, None, merge_operator, None) .await @@ -1401,8 +1401,8 @@ mod tests { batch.merge(b"key1", b"merge3"); // When: extracting entries with a merge operator - let merge_operator = Some(std::sync::Arc::new(StringConcatMergeOperator) - as crate::merge_operator::MergeOperatorType); + let merge_operator = + Some(std::sync::Arc::new(StringConcatMergeOperator) as MergeOperatorType); let (result, _, _) = batch .extract_entries(100, 1000, None, merge_operator, None) .await @@ -1435,8 +1435,8 @@ mod tests { }, ); - let merge_operator = Some(std::sync::Arc::new(StringConcatMergeOperator) - as crate::merge_operator::MergeOperatorType); + let merge_operator = + Some(std::sync::Arc::new(StringConcatMergeOperator) as MergeOperatorType); let (result, _, _) = batch .extract_entries(100, 1000, None, merge_operator, None) .await @@ -1469,8 +1469,8 @@ mod tests { }, ); - let merge_operator = Some(std::sync::Arc::new(StringConcatMergeOperator) - as crate::merge_operator::MergeOperatorType); + let merge_operator = + Some(std::sync::Arc::new(StringConcatMergeOperator) as MergeOperatorType); let err = batch .extract_entries(100, 1000, None, merge_operator, None) .await @@ -1508,8 +1508,8 @@ mod tests { }, ); - let merge_operator = Some(std::sync::Arc::new(StringConcatMergeOperator) - as crate::merge_operator::MergeOperatorType); + let merge_operator = + Some(std::sync::Arc::new(StringConcatMergeOperator) as MergeOperatorType); let (result, _, _) = batch .extract_entries(100, 1000, None, merge_operator, None) .await @@ -1534,8 +1534,8 @@ mod tests { }, ); - let merge_operator = Some(std::sync::Arc::new(StringConcatMergeOperator) - as crate::merge_operator::MergeOperatorType); + let merge_operator = + Some(std::sync::Arc::new(StringConcatMergeOperator) as MergeOperatorType); let err = batch .extract_entries(100, 1000, None, merge_operator, None) .await @@ -1556,8 +1556,8 @@ mod tests { }, ); - let merge_operator = Some(std::sync::Arc::new(StringConcatMergeOperator) - as crate::merge_operator::MergeOperatorType); + let merge_operator = + Some(std::sync::Arc::new(StringConcatMergeOperator) as MergeOperatorType); let err = batch .extract_entries(100, 1000, None, merge_operator, None) .await @@ -1575,8 +1575,8 @@ mod tests { batch.merge(b"key1", b"merge2"); // When: extracting entries with a merge operator - let merge_operator = Some(std::sync::Arc::new(StringConcatMergeOperator) - as crate::merge_operator::MergeOperatorType); + let merge_operator = + Some(std::sync::Arc::new(StringConcatMergeOperator) as MergeOperatorType); let (result, _, _) = batch .extract_entries(100, 1000, None, merge_operator, None) .await @@ -1601,8 +1601,8 @@ mod tests { batch.merge(b"key1", b"merge2"); // When: extracting entries with a merge operator - let merge_operator = Some(std::sync::Arc::new(StringConcatMergeOperator) - as crate::merge_operator::MergeOperatorType); + let merge_operator = + Some(std::sync::Arc::new(StringConcatMergeOperator) as MergeOperatorType); let (result, _, _) = batch .extract_entries(100, 1000, None, merge_operator, None) .await diff --git a/slatedb/src/batch_write.rs b/slatedb/src/batch_write.rs index df264c88d0..13c2f79731 100644 --- a/slatedb/src/batch_write.rs +++ b/slatedb/src/batch_write.rs @@ -391,7 +391,7 @@ impl DbInner { &self, freeze_memtable: bool, ) -> Result<(), SlateDBError> { - let (done, rx) = tokio::sync::oneshot::channel(); + let (done, rx) = oneshot::channel(); self.write_notifier .send(BatchWriterMessage::Flush(BatchWriterFlush { freeze_memtable, @@ -568,11 +568,8 @@ mod tests { fn test_message( batch: WriteBatch, options: WriteOptions, - ) -> ( - BatchWriterMessage, - tokio::sync::oneshot::Receiver, - ) { - let (done, rx) = tokio::sync::oneshot::channel(); + ) -> (BatchWriterMessage, oneshot::Receiver) { + let (done, rx) = oneshot::channel(); ( BatchWriterMessage::WriteBatch(WriteBatchRequest { batch, @@ -623,7 +620,7 @@ mod tests { .unwrap(); let wal_writer = Box::new(FailingWalWriter::new(FailingWalOperation::Flush)); let mut handler = WriteBatchEventHandler::new(db.inner.clone(), Some(wal_writer)); - let (done, done_rx) = tokio::sync::oneshot::channel(); + let (done, done_rx) = oneshot::channel(); let msg = BatchWriterMessage::Flush(BatchWriterFlush { freeze_memtable: false, done, diff --git a/slatedb/src/bytes_generator.rs b/slatedb/src/bytes_generator.rs index 07d9c85b9a..fb746f4c28 100644 --- a/slatedb/src/bytes_generator.rs +++ b/slatedb/src/bytes_generator.rs @@ -30,7 +30,7 @@ impl OrderedBytesGenerator { } pub(crate) fn next(&mut self) -> Bytes { - let mut result = BytesMut::with_capacity(self.bytes.len() + std::mem::size_of::()); + let mut result = BytesMut::with_capacity(self.bytes.len() + size_of::()); result.put_slice(self.bytes.as_slice()); result.put(self.suffix.as_ref()); self.increment(); diff --git a/slatedb/src/bytes_range.rs b/slatedb/src/bytes_range.rs index cbda899f42..3d3a318b7d 100644 --- a/slatedb/src/bytes_range.rs +++ b/slatedb/src/bytes_range.rs @@ -21,69 +21,69 @@ pub trait ByteRangeBounds { fn bound_as_bytes>(bound: Bound<&K>) -> Bound<&[u8]> { match bound { - Bound::Included(k) => Bound::Included(k.as_ref()), - Bound::Excluded(k) => Bound::Excluded(k.as_ref()), - Bound::Unbounded => Bound::Unbounded, + Included(k) => Included(k.as_ref()), + Excluded(k) => Excluded(k.as_ref()), + Unbounded => Unbounded, } } impl> ByteRangeBounds for Range { fn start_bound(&self) -> Bound<&[u8]> { - Bound::Included(self.start.as_ref()) + Included(self.start.as_ref()) } fn end_bound(&self) -> Bound<&[u8]> { - Bound::Excluded(self.end.as_ref()) + Excluded(self.end.as_ref()) } } impl> ByteRangeBounds for RangeInclusive { fn start_bound(&self) -> Bound<&[u8]> { - bound_as_bytes(std::ops::RangeBounds::start_bound(self)) + bound_as_bytes(RangeBounds::start_bound(self)) } fn end_bound(&self) -> Bound<&[u8]> { - bound_as_bytes(std::ops::RangeBounds::end_bound(self)) + bound_as_bytes(RangeBounds::end_bound(self)) } } impl> ByteRangeBounds for RangeFrom { fn start_bound(&self) -> Bound<&[u8]> { - Bound::Included(self.start.as_ref()) + Included(self.start.as_ref()) } fn end_bound(&self) -> Bound<&[u8]> { - Bound::Unbounded + Unbounded } } impl> ByteRangeBounds for RangeTo { fn start_bound(&self) -> Bound<&[u8]> { - Bound::Unbounded + Unbounded } fn end_bound(&self) -> Bound<&[u8]> { - Bound::Excluded(self.end.as_ref()) + Excluded(self.end.as_ref()) } } impl> ByteRangeBounds for RangeToInclusive { fn start_bound(&self) -> Bound<&[u8]> { - Bound::Unbounded + Unbounded } fn end_bound(&self) -> Bound<&[u8]> { - Bound::Included(self.end.as_ref()) + Included(self.end.as_ref()) } } impl ByteRangeBounds for RangeFull { fn start_bound(&self) -> Bound<&[u8]> { - Bound::Unbounded + Unbounded } fn end_bound(&self) -> Bound<&[u8]> { - Bound::Unbounded + Unbounded } } @@ -100,17 +100,17 @@ impl ByteRangeBounds for BytesRange { impl> ByteRangeBounds for (Bound, Bound) { fn start_bound(&self) -> Bound<&[u8]> { match &self.0 { - Bound::Included(v) => Bound::Included(v.as_ref()), - Bound::Excluded(v) => Bound::Excluded(v.as_ref()), - Bound::Unbounded => Bound::Unbounded, + Included(v) => Included(v.as_ref()), + Excluded(v) => Excluded(v.as_ref()), + Unbounded => Unbounded, } } fn end_bound(&self) -> Bound<&[u8]> { match &self.1 { - Bound::Included(v) => Bound::Included(v.as_ref()), - Bound::Excluded(v) => Bound::Excluded(v.as_ref()), - Bound::Unbounded => Bound::Unbounded, + Included(v) => Included(v.as_ref()), + Excluded(v) => Excluded(v.as_ref()), + Unbounded => Unbounded, } } } @@ -301,7 +301,7 @@ impl BytesRange { pub(crate) fn as_point(&self) -> Option<&Bytes> { match (RangeBounds::start_bound(self), RangeBounds::end_bound(self)) { - (Bound::Included(start), Bound::Included(end)) if start == end => Some(start), + (Included(start), Included(end)) if start == end => Some(start), _ => None, } } @@ -315,8 +315,7 @@ pub(crate) mod tests { use bytes::Bytes; use proptest::{prop_assert, prop_assert_eq, proptest}; - use std::ops::Bound; - use std::ops::Bound::{Included, Unbounded}; + use std::ops::Bound::{Excluded, Included, Unbounded}; use std::ops::RangeBounds; #[test] @@ -348,22 +347,17 @@ pub(crate) mod tests { fn test_byte_range_bounds_for_common_shapes() { let full = BytesRange::from_prefix_and_subrange(b"p", ..); assert_eq!(full.start_bound(), Included(&Bytes::from_static(b"p"))); - assert_eq!(full.end_bound(), Bound::Excluded(&Bytes::from_static(b"q"))); + assert_eq!(full.end_bound(), Excluded(&Bytes::from_static(b"q"))); let range = BytesRange::from_prefix_and_subrange(b"", b"a".to_vec()..=b"b".to_vec()); assert_eq!(range.start_bound(), Included(&Bytes::from_static(b"a"))); assert_eq!(range.end_bound(), Included(&Bytes::from_static(b"b"))); - let tuple = BytesRange::from_prefix_and_subrange( - b"ab", - (Bound::Excluded(&b"x"[..]), Bound::Included(&b"y"[..])), - ); + let tuple = + BytesRange::from_prefix_and_subrange(b"ab", (Excluded(&b"x"[..]), Included(&b"y"[..]))); assert_eq!( tuple, - BytesRange::from_prefix_and_subrange( - b"ab", - (Bound::Excluded(&b"x"[..]), Bound::Included(&b"y"[..])) - ) + BytesRange::from_prefix_and_subrange(b"ab", (Excluded(&b"x"[..]), Included(&b"y"[..]))) ); } @@ -372,8 +366,8 @@ pub(crate) mod tests { let range = BytesRange::from_prefix_and_subrange(b"user1:", &b"0005"[..]..&b"0042"[..]); let start = Bytes::from("user1:0005"); let end = Bytes::from("user1:0042"); - assert_eq!(range.start_bound(), Bound::Included(&start)); - assert_eq!(range.end_bound(), Bound::Excluded(&end)); + assert_eq!(range.start_bound(), Included(&start)); + assert_eq!(range.end_bound(), Excluded(&end)); } #[test] @@ -381,8 +375,8 @@ pub(crate) mod tests { let range = BytesRange::from_prefix_and_subrange(b"ab", ..=&b"x"[..]); let start = Bytes::from("ab"); let end = Bytes::from("abx"); - assert_eq!(range.start_bound(), Bound::Included(&start)); - assert_eq!(range.end_bound(), Bound::Included(&end)); + assert_eq!(range.start_bound(), Included(&start)); + assert_eq!(range.end_bound(), Included(&end)); } #[test] @@ -391,18 +385,15 @@ pub(crate) mod tests { let range = BytesRange::from_prefix_and_subrange(b"ab", &b"x"[..]..); let start = Bytes::from("abx"); let end = Bytes::from("ac"); - assert_eq!(range.start_bound(), Bound::Included(&start)); - assert_eq!(range.end_bound(), Bound::Excluded(&end)); + assert_eq!(range.start_bound(), Included(&start)); + assert_eq!(range.end_bound(), Excluded(&end)); } #[test] fn test_from_prefix_and_subrange_excluded_start() { - let range = BytesRange::from_prefix_and_subrange( - b"ab", - (Bound::Excluded(&b"x"[..]), Bound::Unbounded), - ); + let range = BytesRange::from_prefix_and_subrange(b"ab", (Excluded(&b"x"[..]), Unbounded)); let start = Bytes::from("abx"); - assert_eq!(range.start_bound(), Bound::Excluded(&start)); + assert_eq!(range.start_bound(), Excluded(&start)); } #[test] @@ -410,8 +401,8 @@ pub(crate) mod tests { let prefix = vec![0xff, 0xff]; let range = BytesRange::from_prefix_and_subrange(&prefix, &b"a"[..]..); let start = Bytes::from(vec![0xff, 0xff, b'a']); - assert_eq!(range.start_bound(), Bound::Included(&start)); - assert_eq!(range.end_bound(), Bound::Unbounded); + assert_eq!(range.start_bound(), Included(&start)); + assert_eq!(range.end_bound(), Unbounded); } #[test] @@ -453,8 +444,8 @@ pub(crate) mod tests { let range = BytesRange::from_prefix(b"ab"); let start = Bytes::from("ab"); let end = Bytes::from("ac"); - assert_eq!(range.start_bound(), Bound::Included(&start)); - assert_eq!(range.end_bound(), Bound::Excluded(&end)); + assert_eq!(range.start_bound(), Included(&start)); + assert_eq!(range.end_bound(), Excluded(&end)); } #[test] @@ -462,15 +453,15 @@ pub(crate) mod tests { let prefix = vec![0xff, 0xff]; let range = BytesRange::from_prefix(&prefix); let start = Bytes::from(prefix); - assert_eq!(range.start_bound(), Bound::Included(&start)); - assert_eq!(range.end_bound(), Bound::Unbounded); + assert_eq!(range.start_bound(), Included(&start)); + assert_eq!(range.end_bound(), Unbounded); } #[test] fn test_from_prefix_allows_empty_prefix() { let range = BytesRange::from_prefix(b""); - assert_eq!(range.start_bound(), Bound::Unbounded); - assert_eq!(range.end_bound(), Bound::Unbounded); + assert_eq!(range.start_bound(), Unbounded); + assert_eq!(range.end_bound(), Unbounded); } #[test] @@ -567,14 +558,14 @@ pub(crate) mod tests { #[test] #[should_panic(expected = "Range must be non-empty")] fn test_new_with_start_larger_than_end_panics() { - let start = Bound::Included(Bytes::from("z")); - let end = Bound::Included(Bytes::from("a")); + let start = Included(Bytes::from("z")); + let end = Included(Bytes::from("a")); BytesRange::new(start, end); } #[test] fn test_empty_included_start_bound_is_valid_and_contains_all_keys() { - let range = BytesRange::new(Bound::Included(Bytes::new()), Bound::Unbounded); + let range = BytesRange::new(Included(Bytes::new()), Unbounded); assert!(range.contains(&Bytes::new())); // b"" <= b"" holds for Included assert!(range.contains(&Bytes::from("a"))); assert!(range.contains(&Bytes::from("z"))); @@ -582,7 +573,7 @@ pub(crate) mod tests { #[test] fn test_empty_excluded_start_bound_is_valid_and_contains_all_keys() { - let range = BytesRange::new(Bound::Excluded(Bytes::new()), Bound::Unbounded); + let range = BytesRange::new(Excluded(Bytes::new()), Unbounded); assert!(!range.contains(&Bytes::new())); // b"" < b"" is false for Excluded assert!(range.contains(&Bytes::from("a"))); assert!(range.contains(&Bytes::from("z"))); diff --git a/slatedb/src/cached_object_store/object_store.rs b/slatedb/src/cached_object_store/object_store.rs index 216a408a38..8ffc9f82e9 100644 --- a/slatedb/src/cached_object_store/object_store.rs +++ b/slatedb/src/cached_object_store/object_store.rs @@ -336,8 +336,8 @@ impl CachedObjectStore { async fn cached_put_opts( &self, location: &Path, - payload: object_store::PutPayload, - opts: object_store::PutOptions, + payload: PutPayload, + opts: PutOptions, ) -> object_store::Result { // The per-call tag decides whether this write is cached. let tag = ObjectStoreCallTag::from_extensions(&opts.extensions); @@ -1406,7 +1406,7 @@ mod tests { inner .put( &location, - PutPayload::from_bytes(bytes::Bytes::from_static(b"hello world")), + PutPayload::from_bytes(Bytes::from_static(b"hello world")), ) .await .unwrap(); @@ -1426,10 +1426,7 @@ mod tests { .expect("cache miss should fetch from inner store"); assert!(result.extensions.get::().is_some()); - assert_eq!( - result.bytes().await.unwrap(), - bytes::Bytes::from_static(b"hello") - ); + assert_eq!(result.bytes().await.unwrap(), Bytes::from_static(b"hello")); } #[tokio::test] @@ -1439,7 +1436,7 @@ mod tests { inner .put( &location, - PutPayload::from_bytes(bytes::Bytes::from_static(b"hello")), + PutPayload::from_bytes(Bytes::from_static(b"hello")), ) .await .unwrap(); diff --git a/slatedb/src/cached_object_store/storage_fs.rs b/slatedb/src/cached_object_store/storage_fs.rs index 8a1500e533..8ace3f99d1 100644 --- a/slatedb/src/cached_object_store/storage_fs.rs +++ b/slatedb/src/cached_object_store/storage_fs.rs @@ -194,11 +194,7 @@ impl FsCacheStorage { #[async_trait::async_trait] impl LocalCacheStorage for FsCacheStorage { - fn entry( - &self, - location: &object_store::path::Path, - part_size: usize, - ) -> Box { + fn entry(&self, location: &Path, part_size: usize) -> Box { Box::new(FsCacheEntry { root_folder: self.root_folder.clone(), location: location.clone(), @@ -1073,7 +1069,7 @@ impl FsCacheEvictorInner { // a specific index. Returns None if no available index exists. fn pick_random_available_index( &self, - rng: &mut impl rand::Rng, + rng: &mut impl Rng, keys: &[std::path::PathBuf], picked: &HashSet, exclude_idx: Option, @@ -1197,7 +1193,7 @@ mod tests { .prefix("objstore_cache_test_evictor_") .tempdir() .unwrap(); - let recorder = slatedb_common::metrics::MetricsRecorderHelper::noop(); + let recorder = MetricsRecorderHelper::noop(); let evictor = FsCacheEvictorInner::new( temp_dir.path().to_path_buf(), @@ -1225,7 +1221,7 @@ mod tests { .await; assert_eq!(evicted, 2048); - let file_paths = walkdir::WalkDir::new(temp_dir.path()) + let file_paths = WalkDir::new(temp_dir.path()) .into_iter() .map(|entry| entry.unwrap().file_name().to_string_lossy().to_string()) .collect::>(); @@ -1238,7 +1234,7 @@ mod tests { .prefix("objstore_cache_test_evictor_backpressure_") .tempdir() .unwrap(); - let recorder = slatedb_common::metrics::MetricsRecorderHelper::noop(); + let recorder = MetricsRecorderHelper::noop(); let evictor = FsCacheEvictor::new( temp_dir.path().to_path_buf(), @@ -1275,7 +1271,7 @@ mod tests { .prefix("objstore_cache_test_evictor_") .tempdir() .unwrap(); - let recorder = slatedb_common::metrics::MetricsRecorderHelper::noop(); + let recorder = MetricsRecorderHelper::noop(); let evictor = Arc::new(FsCacheEvictorInner::new( temp_dir.path().to_path_buf(), 1024 * 2, @@ -1306,7 +1302,7 @@ mod tests { .prefix("objstore_cache_test_evictor_") .tempdir() .unwrap(); - let recorder = slatedb_common::metrics::MetricsRecorderHelper::noop(); + let recorder = MetricsRecorderHelper::noop(); let evictor = Arc::new(FsCacheEvictorInner::new( temp_dir.path().to_path_buf(), @@ -1358,7 +1354,7 @@ mod tests { .prefix("objstore_cache_test_pick_") .tempdir() .unwrap(); - let recorder = slatedb_common::metrics::MetricsRecorderHelper::noop(); + let recorder = MetricsRecorderHelper::noop(); let evictor = FsCacheEvictorInner::new( temp_dir.path().to_path_buf(), 1024, diff --git a/slatedb/src/clone.rs b/slatedb/src/clone.rs index d5092c88ca..852be58907 100644 --- a/slatedb/src/clone.rs +++ b/slatedb/src/clone.rs @@ -1112,7 +1112,7 @@ mod tests { // Create an uninitialized manifest with an invalid checkpoint id let clone_manifest_store = Arc::new(ManifestStore::new(&clone_path, object_store.clone())); - let non_existent_source_checkpoint_id = uuid::Uuid::new_v4(); + let non_existent_source_checkpoint_id = Uuid::new_v4(); StoredManifest::store_uninitialized_clone( clone_manifest_store, Manifest::cloned( @@ -1226,7 +1226,7 @@ mod tests { Manifest::cloned( &parent_manifest, original_parent_path.to_string(), - uuid::Uuid::new_v4(), + Uuid::new_v4(), rand.clone(), ), system_clock.clone(), diff --git a/slatedb/src/compaction_execute_bench.rs b/slatedb/src/compaction_execute_bench.rs index d5e145bb88..3ee197b89f 100644 --- a/slatedb/src/compaction_execute_bench.rs +++ b/slatedb/src/compaction_execute_bench.rs @@ -1,5 +1,4 @@ use std::collections::HashMap; -use std::mem; use std::sync::Arc; use std::time::Duration; @@ -83,7 +82,7 @@ impl CompactionExecuteBench { BlockCachePolicy::default(), )); let num_keys = sst_bytes / (val_bytes + key_bytes); - let mut key_start = vec![0u8; key_bytes - mem::size_of::()]; + let mut key_start = vec![0u8; key_bytes - size_of::()]; self.rand.rng().fill_bytes(key_start.as_mut_slice()); let mut futures = FuturesUnordered::>>::new(); for i in 0..num_ssts { diff --git a/slatedb/src/compaction_worker.rs b/slatedb/src/compaction_worker.rs index a330d31b39..836e0c4aa6 100644 --- a/slatedb/src/compaction_worker.rs +++ b/slatedb/src/compaction_worker.rs @@ -1376,13 +1376,13 @@ mod tests { let (tx, rx) = async_channel::unbounded::(); let executor: Arc = Arc::new( TokioCompactionExecutor::new(TokioCompactionExecutorOptions { - handle: tokio::runtime::Handle::current(), + handle: Handle::current(), options: options.clone(), worker_tx: tx, table_store: table_store.clone(), rand: Arc::new(DbRand::new(100u64)), stats: { - let recorder = slatedb_common::metrics::MetricsRecorderHelper::noop(); + let recorder = MetricsRecorderHelper::noop(); Arc::new(CompactionStats::new(&recorder)) }, worker_stats: WorkerStats::noop(), @@ -1428,7 +1428,7 @@ mod tests { COMPACTION_WORKER_TASK_NAME.to_string(), Box::new(handler), rx, - &tokio::runtime::Handle::current(), + &Handle::current(), ) .unwrap(); let worker = CompactionWorker::new(task_executor); diff --git a/slatedb/src/compactions_store.rs b/slatedb/src/compactions_store.rs index e264b244a7..45e4c37361 100644 --- a/slatedb/src/compactions_store.rs +++ b/slatedb/src/compactions_store.rs @@ -276,7 +276,6 @@ impl CompactionsStore { mod tests { use super::*; use crate::compactor_state::{Compaction, CompactionSpec, SourceId}; - use crate::error; use crate::retrying_object_store::RetryingObjectStore; use crate::test_utils::FlakyObjectStore; use object_store::memory::InMemory; @@ -301,7 +300,7 @@ mod tests { assert!(matches!( result.unwrap_err(), - error::SlateDBError::TransactionalObjectVersionExists + SlateDBError::TransactionalObjectVersionExists )); } @@ -395,7 +394,7 @@ mod tests { .unwrap(); let result = compactor1.refresh().await; - assert!(matches!(result, Err(error::SlateDBError::Fenced))); + assert!(matches!(result, Err(SlateDBError::Fenced))); } #[tokio::test] diff --git a/slatedb/src/compactor.rs b/slatedb/src/compactor.rs index 7806b61bab..b36c2a2c54 100644 --- a/slatedb/src/compactor.rs +++ b/slatedb/src/compactor.rs @@ -188,7 +188,7 @@ pub trait CompactionScheduler: Send + Sync { "rejected full-segment compaction: unknown segment {:?}", segment ); - return Err(crate::Error::from(SlateDBError::InvalidCompaction)); + return Err(Error::from(SlateDBError::InvalidCompaction)); }; match plan_full_tree(segment, tree) { Some(spec) => Ok(vec![spec]), @@ -200,7 +200,7 @@ pub trait CompactionScheduler: Send + Sync { segment ); } - Err(crate::Error::from(SlateDBError::InvalidCompaction)) + Err(Error::from(SlateDBError::InvalidCompaction)) } } } @@ -373,12 +373,12 @@ impl Compactor { /// /// ## Returns /// - `Ok(())` when the compactor task exits cleanly, or [`SlateDBError`] on failure. - pub async fn run(&self) -> Result<(), crate::Error> { + pub async fn run(&self) -> Result<(), Error> { self.start().await?; self.join().await } - pub(crate) async fn start(&self) -> Result<(), crate::Error> { + pub(crate) async fn start(&self) -> Result<(), Error> { // The coordinator delegates compaction execution to [`crate::compaction_worker::CompactionWorker`] // either spawned in this process (set `worker: Some`) or running standalone (set `worker: None`). let (_tx, rx) = async_channel::unbounded::(); @@ -401,7 +401,7 @@ impl Compactor { rx, &Handle::current(), ) - .map_err(crate::Error::from)?; + .map_err(Error::from)?; // Spawn an in-process worker if configured. The worker runs under its // own cancellation token; Compactor::stop and run() are responsible for @@ -429,23 +429,23 @@ impl Compactor { worker_rx, &Handle::current(), ) - .map_err(crate::Error::from)?; + .map_err(Error::from)?; } self.task_executor.monitor_on(&Handle::current())?; Ok(()) } - pub(crate) async fn join(&self) -> Result<(), crate::Error> { + pub(crate) async fn join(&self) -> Result<(), Error> { self.task_executor .join_task(COMPACTOR_TASK_NAME) .await - .map_err(crate::Error::from)?; + .map_err(Error::from)?; if self.options.worker.is_some() { self.task_executor .join_task(crate::compaction_worker::COMPACTION_WORKER_TASK_NAME) .await - .map_err(crate::Error::from)?; + .map_err(Error::from)?; } Ok(()) } @@ -454,16 +454,16 @@ impl Compactor { /// /// ## Returns /// - `Ok(())` once the task has shut down, or [`SlateDBError`] if shutdown fails. - pub async fn stop(&self) -> Result<(), crate::Error> { + pub async fn stop(&self) -> Result<(), Error> { self.task_executor .shutdown_task(COMPACTOR_TASK_NAME) .await - .map_err(crate::Error::from)?; + .map_err(Error::from)?; if self.options.worker.is_some() { self.task_executor .shutdown_task(crate::compaction_worker::COMPACTION_WORKER_TASK_NAME) .await - .map_err(crate::Error::from)?; + .map_err(Error::from)?; } Ok(()) } @@ -480,13 +480,13 @@ impl Compactor { compactions_store: Arc, rand: Arc, system_clock: Arc, - ) -> Result { + ) -> Result { let compaction_id = rand.rng().gen_ulid(system_clock.as_ref()); let compaction = Compaction::new(compaction_id, spec); let mut stored_compactions = match StoredCompactions::try_load(compactions_store.clone()).await? { Some(stored) => stored, - None => return Err(crate::Error::from(SlateDBError::InvalidDBState)), + None => return Err(Error::from(SlateDBError::InvalidDBState)), }; loop { @@ -497,7 +497,7 @@ impl Compactor { Err(err) if err.is_sequenced_write_conflict() => { stored_compactions.refresh().await?; } - Err(err) => return Err(crate::Error::from(err)), + Err(err) => return Err(Error::from(err)), } } } @@ -972,16 +972,12 @@ impl CompactorEventHandler { ); return Err(SlateDBError::InvalidCompaction); }; - let l0_view_ids = tree - .l0 - .iter() - .map(|view| view.id) - .collect::>(); + let l0_view_ids = tree.l0.iter().map(|view| view.id).collect::>(); let sr_ids = tree .compacted .iter() .map(|sr| sr.id) - .collect::>(); + .collect::>(); if let Some(missing) = spec.sources().iter().find(|source| match source { SourceId::SstView(id) => !l0_view_ids.contains(id), @@ -1107,7 +1103,7 @@ impl CompactorEventHandler { if !compaction.is_drain() { return Ok(()); } - let drained_l0_ids: std::collections::HashSet = compaction + let drained_l0_ids: HashSet = compaction .sources() .iter() .filter_map(|s| match s { @@ -3121,7 +3117,7 @@ mod tests { db.merge_with_options( b"key1", &[b'a'; 32], - &crate::config::MergeOptions { + &MergeOptions { ttl: Ttl::ExpireAfterMillis(10), }, &WriteOptions { @@ -3139,7 +3135,7 @@ mod tests { db.merge_with_options( b"key1", &[b'b'; 32], - &crate::config::MergeOptions { ttl: Ttl::NoExpiry }, + &MergeOptions { ttl: Ttl::NoExpiry }, &WriteOptions { ..Default::default() }, @@ -3312,7 +3308,7 @@ mod tests { db.merge_with_options( b"key1", b"a", - &crate::config::MergeOptions { + &MergeOptions { ttl: Ttl::ExpireAfterMillis(100), }, &WriteOptions { @@ -3336,7 +3332,7 @@ mod tests { db.merge_with_options( b"key1", b"b", - &crate::config::MergeOptions { + &MergeOptions { ttl: Ttl::ExpireAfterMillis(200), }, &WriteOptions { @@ -5415,7 +5411,7 @@ mod tests { // Build the handler and trigger a ticker to pick up the pre-existing Submitted entry. let scheduler = Arc::new(MockScheduler::new()); let rand = Arc::new(DbRand::default()); - let recorder = slatedb_common::metrics::MetricsRecorderHelper::noop(); + let recorder = MetricsRecorderHelper::noop(); let compactor_stats = Arc::new(CompactionStats::new(&recorder)); let mut handler = CompactorEventHandler::new( manifest_store, @@ -6586,11 +6582,9 @@ mod tests { assert!(c.worker().is_none(), "worker should be cleared"); // and: the reclamation is counted. - let reclaimed = slatedb_common::metrics::lookup_metric( - &fixture.test_recorder, - crate::compactor::stats::JOBS_RECLAIMED, - ) - .expect("metric not found"); + let reclaimed = + slatedb_common::metrics::lookup_metric(&fixture.test_recorder, stats::JOBS_RECLAIMED) + .expect("metric not found"); assert_eq!(reclaimed, 1, "one job should be counted as reclaimed"); } @@ -6602,11 +6596,8 @@ mod tests { let mut fixture = CompactorEventHandlerTestFixture::new().await; let claimed_count = || { - slatedb_common::metrics::lookup_metric( - &fixture.test_recorder, - crate::compactor::stats::JOBS_CLAIMED, - ) - .expect("metric not found") + slatedb_common::metrics::lookup_metric(&fixture.test_recorder, stats::JOBS_CLAIMED) + .expect("metric not found") }; // given: a job already claimed (Running) before the coordinator's first tick. diff --git a/slatedb/src/compactor_state.rs b/slatedb/src/compactor_state.rs index 4e5382cfdd..4c09ef1df9 100644 --- a/slatedb/src/compactor_state.rs +++ b/slatedb/src/compactor_state.rs @@ -2120,7 +2120,7 @@ mod tests { let original_l0s = &state.db_state().clone().tree.l0; let original_srs = &state.db_state().clone().tree.compacted; // L0: from 4th onward (index > 2) - let l0_sources = original_l0s.iter().skip(3).map(|h| SourceId::SstView(h.id)); + let l0_sources = original_l0s.iter().skip(3).map(|h| SstView(h.id)); // SRs: first 3 (index < 3) let sr_sources = original_srs @@ -2212,7 +2212,7 @@ mod tests { } fn build_l0_compaction(ssts: &VecDeque, dst: u32) -> CompactionSpec { - let sources = ssts.iter().map(|h| SourceId::SstView(h.id)).collect(); + let sources = ssts.iter().map(|h| SstView(h.id)).collect(); CompactionSpec::new(sources, dst) } @@ -2398,13 +2398,13 @@ mod tests { }]; let first_id = rand.rng().gen_ulid(system_clock.as_ref()); - let first = CompactionSpec::drain_segment(prefix.clone(), vec![SourceId::SstView(l0_a.id)]); + let first = CompactionSpec::drain_segment(prefix.clone(), vec![SstView(l0_a.id)]); state .add_compaction(Compaction::new(first_id, first)) .expect("first drain must register"); let second_id = rand.rng().gen_ulid(system_clock.as_ref()); - let second = CompactionSpec::drain_segment(prefix, vec![SourceId::SstView(l0_b.id)]); + let second = CompactionSpec::drain_segment(prefix, vec![SstView(l0_b.id)]); let err = state .add_compaction(Compaction::new(second_id, second)) .expect_err("second drain on same segment must be rejected"); @@ -2444,13 +2444,13 @@ mod tests { ]; let first_id = rand.rng().gen_ulid(system_clock.as_ref()); - let first = CompactionSpec::drain_segment(prefix_a, vec![SourceId::SstView(l0_a.id)]); + let first = CompactionSpec::drain_segment(prefix_a, vec![SstView(l0_a.id)]); state .add_compaction(Compaction::new(first_id, first)) .expect("first drain must register"); let second_id = rand.rng().gen_ulid(system_clock.as_ref()); - let second = CompactionSpec::drain_segment(prefix_b, vec![SourceId::SstView(l0_b.id)]); + let second = CompactionSpec::drain_segment(prefix_b, vec![SstView(l0_b.id)]); state .add_compaction(Compaction::new(second_id, second)) .expect("drain on a different segment must register"); @@ -2476,7 +2476,7 @@ mod tests { }]; let compaction_id = rand.rng().gen_ulid(system_clock.as_ref()); - let spec = CompactionSpec::drain_segment(prefix, vec![SourceId::SstView(l0.id)]); + let spec = CompactionSpec::drain_segment(prefix, vec![SstView(l0.id)]); state .add_compaction(Compaction::new(compaction_id, spec)) .expect("drain submission must register"); @@ -2531,8 +2531,8 @@ mod tests { let spec = CompactionSpec::drain_segment( prefix.clone(), vec![ - SourceId::SstView(l0_newer.id), - SourceId::SstView(l0_older.id), + SstView(l0_newer.id), + SstView(l0_older.id), SourceId::SortedRun(sr.id), ], ); @@ -2579,8 +2579,7 @@ mod tests { }]; let compaction_id = rand.rng().gen_ulid(system_clock.as_ref()); - let spec = - CompactionSpec::drain_segment(prefix.clone(), vec![SourceId::SstView(l0_observed.id)]); + let spec = CompactionSpec::drain_segment(prefix.clone(), vec![SstView(l0_observed.id)]); state .add_compaction(Compaction::new(compaction_id, spec)) .expect("drain compaction must register"); diff --git a/slatedb/src/config.rs b/slatedb/src/config.rs index 5a2a9537ae..27339db7ab 100644 --- a/slatedb/src/config.rs +++ b/slatedb/src/config.rs @@ -1052,7 +1052,7 @@ impl Settings { } impl Provider for Settings { - fn metadata(&self) -> figment::Metadata { + fn metadata(&self) -> Metadata { Metadata::named("SlateDb configuration options") } diff --git a/slatedb/src/db.rs b/slatedb/src/db.rs index 7b32750955..3ba6f5967b 100644 --- a/slatedb/src/db.rs +++ b/slatedb/src/db.rs @@ -133,7 +133,7 @@ impl DbInner { wal_observer: Box, recorder: MetricsRecorderHelper, fp_registry: Arc, - merge_operator: Option, + merge_operator: Option, status_manager: Arc, segment_extractor: Option>, ) -> Result { @@ -2724,7 +2724,7 @@ mod tests { kv_store.close().await.unwrap(); } - fn assert_value(entry: &crate::types::RowEntry, expected: &[u8]) { + fn assert_value(entry: &RowEntry, expected: &[u8]) { match &entry.value { crate::types::ValueDeletable::Value(v) => assert_eq!(v.as_ref(), expected), other => panic!("expected Value({expected:?}), got {other:?}"), @@ -3349,7 +3349,7 @@ mod tests { // Simulate a failed state (e.g. fenced). db.inner .status_manager - .write_result(Err(crate::error::SlateDBError::Fenced)); + .write_result(Err(SlateDBError::Fenced)); // close() should succeed but not flush when failed. db.close().await.unwrap(); @@ -7625,9 +7625,7 @@ mod tests { // freeze the job after its first output SSTs upload. Manifest and // `.compactions` I/O use the ungated store passed to `Db::builder`, // so the worker's heartbeats keep flowing while the job is frozen. - let gated = Arc::new(crate::test_utils::GatedObjectStore::new( - object_store.clone(), - )); + let gated = Arc::new(GatedObjectStore::new(object_store.clone())); let gated_store: Arc = gated.clone(); let db = Db::builder(path, object_store.clone()) @@ -7791,19 +7789,13 @@ mod tests { } db.put(b"key1", b"value1").await.unwrap(); - db.inner - .flush_memtables(crate::memtable_flusher::FlushTarget::All) - .await - .unwrap(); + db.inner.flush_memtables(FlushTarget::All).await.unwrap(); let txn = db.begin(IsolationLevel::Snapshot).await.unwrap(); let txn_seq = txn.seqnum(); db.put(b"key2", b"value2").await.unwrap(); - db.inner - .flush_memtables(crate::memtable_flusher::FlushTarget::All) - .await - .unwrap(); + db.inner.flush_memtables(FlushTarget::All).await.unwrap(); let min_active_seq = db.inner.txn_manager.min_active_seq(); assert_eq!(min_active_seq, Some(txn_seq)); @@ -7820,10 +7812,7 @@ mod tests { drop(txn); db.put(b"key3", b"value3").await.unwrap(); - db.inner - .flush_memtables(crate::memtable_flusher::FlushTarget::All) - .await - .unwrap(); + db.inner.flush_memtables(FlushTarget::All).await.unwrap(); { let state = db.inner.state.read(); @@ -7844,19 +7833,13 @@ mod tests { let db = Db::builder(path, object_store).build().await.unwrap(); db.put(b"key1", b"value1").await.unwrap(); - db.inner - .flush_memtables(crate::memtable_flusher::FlushTarget::All) - .await - .unwrap(); + db.inner.flush_memtables(FlushTarget::All).await.unwrap(); let snapshot = db.snapshot().await.unwrap(); let snapshot_seq = snapshot.seq(); db.put(b"key2", b"value2").await.unwrap(); - db.inner - .flush_memtables(crate::memtable_flusher::FlushTarget::All) - .await - .unwrap(); + db.inner.flush_memtables(FlushTarget::All).await.unwrap(); let txn = db.begin(IsolationLevel::Snapshot).await.unwrap(); let txn_seq = txn.seqnum(); @@ -7869,10 +7852,7 @@ mod tests { assert!(snapshot_seq < txn_seq); db.put(b"key3", b"value3").await.unwrap(); - db.inner - .flush_memtables(crate::memtable_flusher::FlushTarget::All) - .await - .unwrap(); + db.inner.flush_memtables(FlushTarget::All).await.unwrap(); { let state = db.inner.state.read(); @@ -7887,10 +7867,7 @@ mod tests { drop(snapshot); db.put(b"key4", b"value4").await.unwrap(); - db.inner - .flush_memtables(crate::memtable_flusher::FlushTarget::All) - .await - .unwrap(); + db.inner.flush_memtables(FlushTarget::All).await.unwrap(); assert_eq!(db.inner.snapshot_manager.min_active_seq(), None); assert_eq!(db.inner.txn_manager.min_active_seq(), Some(txn_seq)); @@ -7914,19 +7891,13 @@ mod tests { let db = Db::builder(path, object_store).build().await.unwrap(); db.put(b"key1", b"value1").await.unwrap(); - db.inner - .flush_memtables(crate::memtable_flusher::FlushTarget::All) - .await - .unwrap(); + db.inner.flush_memtables(FlushTarget::All).await.unwrap(); let txn = db.begin(IsolationLevel::Snapshot).await.unwrap(); let txn_seq = txn.seqnum(); db.put(b"key2", b"value2").await.unwrap(); - db.inner - .flush_memtables(crate::memtable_flusher::FlushTarget::All) - .await - .unwrap(); + db.inner.flush_memtables(FlushTarget::All).await.unwrap(); let snapshot = db.snapshot().await.unwrap(); let snapshot_seq = snapshot.seq(); @@ -7939,10 +7910,7 @@ mod tests { assert!(txn_seq < snapshot_seq); db.put(b"key3", b"value3").await.unwrap(); - db.inner - .flush_memtables(crate::memtable_flusher::FlushTarget::All) - .await - .unwrap(); + db.inner.flush_memtables(FlushTarget::All).await.unwrap(); { let state = db.inner.state.read(); @@ -7957,10 +7925,7 @@ mod tests { drop(txn); db.put(b"key4", b"value4").await.unwrap(); - db.inner - .flush_memtables(crate::memtable_flusher::FlushTarget::All) - .await - .unwrap(); + db.inner.flush_memtables(FlushTarget::All).await.unwrap(); assert_eq!(db.inner.txn_manager.min_active_seq(), None); assert_eq!( @@ -9271,7 +9236,7 @@ mod tests { // When: the DB is fenced (simulated via closed_result) db.inner .status_manager - .write_result(Err(crate::error::SlateDBError::Fenced)); + .write_result(Err(SlateDBError::Fenced)); // Then: the watcher should report close_reason = Fenced let status = tokio::time::timeout( @@ -10655,7 +10620,7 @@ mod tests { async fn test_wal_replay_rejects_empty_extractor_prefix() { #[derive(Debug)] struct AliasedAlwaysEmptyExtractor; - impl crate::PrefixExtractor for AliasedAlwaysEmptyExtractor { + impl PrefixExtractor for AliasedAlwaysEmptyExtractor { fn name(&self) -> &str { "fixed-3" } @@ -10874,10 +10839,10 @@ mod tests { } impl ExtractorConfig { - fn to_extractor(self) -> Option> { + fn to_extractor(self) -> Option> { #[derive(Debug)] struct OtherExtractor; - impl crate::PrefixExtractor for OtherExtractor { + impl PrefixExtractor for OtherExtractor { fn name(&self) -> &str { "other" } @@ -11855,12 +11820,12 @@ mod tests { } /// A path under the db root, e.g. `sub_path("wal/00..002.sst")`. - fn sub_path(&self, suffix: &str) -> object_store::path::Path { - object_store::path::Path::from(format!("{}/{}", self.db_path, suffix)) + fn sub_path(&self, suffix: &str) -> Path { + Path::from(format!("{}/{}", self.db_path, suffix)) } /// Number of cached part files for an object. - fn cached_part_count(&self, path: &object_store::path::Path) -> usize { + fn cached_part_count(&self, path: &Path) -> usize { let dir = self.cache_root.join(path.to_string()); let Ok(entries) = std::fs::read_dir(dir) else { return 0; @@ -11876,7 +11841,7 @@ mod tests { .count() } - fn assert_cached(&self, path: &object_store::path::Path, expected_parts: usize) { + fn assert_cached(&self, path: &Path, expected_parts: usize) { assert_eq!( self.cached_part_count(path), expected_parts, @@ -11897,7 +11862,7 @@ mod tests { } /// Lists the compacted SSTs currently in the object store. - async fn compacted_locations(&self) -> Vec { + async fn compacted_locations(&self) -> Vec { let prefix = self.sub_path("compacted"); self.upstream .list(Some(&prefix)) @@ -11907,13 +11872,13 @@ mod tests { } /// The size of an object as stored upstream, in bytes. - async fn object_size(&self, path: &object_store::path::Path) -> u64 { + async fn object_size(&self, path: &Path) -> u64 { self.upstream.head(path).await.unwrap().size } /// The upstream path of a compacted SST id. - fn compacted_sst_path(&self, id: &SsTableId) -> object_store::path::Path { - crate::paths::PathResolver::from_root(self.db_path.as_str()).sst_path(id) + fn compacted_sst_path(&self, id: &SsTableId) -> Path { + PathResolver::from_root(self.db_path.as_str()).sst_path(id) } fn l0_ids(&self) -> Vec { diff --git a/slatedb/src/db/builder.rs b/slatedb/src/db/builder.rs index 22192e0899..aecfbef28d 100644 --- a/slatedb/src/db/builder.rs +++ b/slatedb/src/db/builder.rs @@ -1148,7 +1148,7 @@ pub(crate) struct CompactorHandlers { /// The embedded worker handler and its receiver, present when /// [`CompactorOptions::worker`] is `Some`. pub(crate) worker: Option<( - crate::compaction_worker::CompactionWorkerHandler, + CompactionWorkerHandler, async_channel::Receiver, )>, } @@ -2097,7 +2097,7 @@ impl + Clone> CloneBuilder { r.start_bound().cloned(), r.end_bound().cloned(), ) - .ok_or_else(|| crate::error::SlateDBError::InvalidProjection { + .ok_or_else(|| SlateDBError::InvalidProjection { prefix: Bytes::copy_from_slice(prefix), reason: "empty range".into(), }) diff --git a/slatedb/src/db_cache/mod.rs b/slatedb/src/db_cache/mod.rs index 1831702c8c..e4662e19bd 100644 --- a/slatedb/src/db_cache/mod.rs +++ b/slatedb/src/db_cache/mod.rs @@ -320,7 +320,7 @@ impl From<(SsTableId, u64)> for CachedKey { #[derive(Clone)] pub(crate) struct EncodedCachedFilter { pub(crate) name: String, - pub(crate) data: bytes::Bytes, + pub(crate) data: Bytes, } #[non_exhaustive] @@ -1477,11 +1477,11 @@ mod tests { // given: a cache that always returns errors let recorder = Arc::new(DefaultMetricsRecorder::new()); let helper = MetricsRecorderHelper::new(recorder.clone(), MetricLevel::default()); - let failing_cache: Arc = Arc::new(super::test_utils::FailingCache); - let cache = super::DbCacheWrapper::new( + let failing_cache: Arc = Arc::new(super::test_utils::FailingCache); + let cache = DbCacheWrapper::new( failing_cache, &helper, - Arc::new(slatedb_common::clock::DefaultSystemClock::default()), + Arc::new(DefaultSystemClock::default()), ); let key = CachedKey::from((SST_ID, 12345u64)); diff --git a/slatedb/src/db_cache/serde.rs b/slatedb/src/db_cache/serde.rs index fbe4412d82..dae600f564 100644 --- a/slatedb/src/db_cache/serde.rs +++ b/slatedb/src/db_cache/serde.rs @@ -321,9 +321,9 @@ mod tests { let policy = BloomFilterPolicy::new(10); let mut builder = policy.builder(); for k in [b"foo", b"bar", b"baz"] { - builder.add_entry(&crate::types::RowEntry::new( - bytes::Bytes::copy_from_slice(k), - crate::types::ValueDeletable::Value(bytes::Bytes::new()), + builder.add_entry(&RowEntry::new( + Bytes::copy_from_slice(k), + crate::types::ValueDeletable::Value(Bytes::new()), 0, None, None, diff --git a/slatedb/src/db_iter.rs b/slatedb/src/db_iter.rs index 5246d84b42..7d23afb551 100644 --- a/slatedb/src/db_iter.rs +++ b/slatedb/src/db_iter.rs @@ -269,9 +269,7 @@ impl DbIterator { match entry_opt { Some(entry) => { if entry.value.is_tombstone() { - return Err(crate::Error::from( - crate::error::SlateDBError::UnexpectedTombstone, - )); + return Err(crate::Error::from(SlateDBError::UnexpectedTombstone)); } Ok(Some(KeyValue::from(entry))) } diff --git a/slatedb/src/db_reader.rs b/slatedb/src/db_reader.rs index badc554d78..abe16e28a6 100644 --- a/slatedb/src/db_reader.rs +++ b/slatedb/src/db_reader.rs @@ -871,7 +871,7 @@ impl DbReader { pub(crate) async fn preload_cache( &self, cached_obj_store: &CachedObjectStore, - path: object_store::path::Path, + path: Path, ) -> Result<(), SlateDBError> { let state = Arc::clone(&self.inner.state.read()); let external_ssts = state.manifest.external_ssts(); @@ -1855,7 +1855,7 @@ mod tests { let parent_manifest = Manifest::initial(ManifestCore::new()); let parent_path = "/tmp/parent_store".to_string(); - let source_checkpoint_id = uuid::Uuid::new_v4(); + let source_checkpoint_id = Uuid::new_v4(); let _ = StoredManifest::store_uninitialized_clone( Arc::clone(&manifest_store), @@ -3607,8 +3607,7 @@ mod tests { let mut test_provider = TestProvider::new(path.clone(), Arc::clone(&object_store)); test_provider.system_clock = clock.clone(); - let merge_operator: crate::merge_operator::MergeOperatorType = - Arc::new(crate::test_utils::StringConcatMergeOperator); + let merge_operator: MergeOperatorType = Arc::new(test_utils::StringConcatMergeOperator); let db = Db::builder(path.clone(), Arc::clone(&object_store)) .with_settings(Settings { diff --git a/slatedb/src/db_state.rs b/slatedb/src/db_state.rs index 7e0b6dc856..0731d4cda8 100644 --- a/slatedb/src/db_state.rs +++ b/slatedb/src/db_state.rs @@ -92,8 +92,8 @@ impl SsTableView { /// where no `DbRand` is available and the id is not stored in the manifest. pub(crate) fn identity(sst: SsTableHandle) -> Self { let id = match &sst.id { - SsTableId::Compacted(ulid) => *ulid, - SsTableId::Wal(wal_id) => Ulid::from_parts(*wal_id, 0), + Compacted(ulid) => *ulid, + Wal(wal_id) => Ulid::from_parts(*wal_id, 0), }; Self::new(id, sst) } @@ -385,7 +385,7 @@ impl SsTableId { } impl Debug for SsTableId { - fn fmt(&self, f: &mut std::fmt::Formatter) -> Result<(), std::fmt::Error> { + fn fmt(&self, f: &mut Formatter) -> Result<(), std::fmt::Error> { match self { Wal(id) => write!(f, "SsTableId::Wal({})", id), Compacted(id) => write!(f, "SsTableId::Compacted({})", id.to_string()), @@ -406,8 +406,8 @@ pub enum SstType { impl From<&SsTableId> for SstType { fn from(id: &SsTableId) -> Self { match id { - SsTableId::Wal(_) => SstType::Wal, - SsTableId::Compacted(_) => SstType::Compacted, + Wal(_) => SstType::Wal, + Compacted(_) => SstType::Compacted, } } } @@ -865,8 +865,8 @@ mod tests { use proptest::proptest; use slatedb_common::clock::{DefaultSystemClock, SystemClock}; use std::collections::BTreeSet; - use std::collections::Bound::Included; use std::collections::VecDeque; + use std::ops::Bound::{Excluded, Included, Unbounded}; use std::ops::RangeBounds; use std::sync::Arc; @@ -1275,7 +1275,7 @@ mod tests { #[test] fn max_l0_overlap_empty_is_zero() { - let l0: std::collections::VecDeque = std::collections::VecDeque::new(); + let l0: VecDeque = VecDeque::new(); assert_eq!(super::max_l0_overlap(&l0), 0); } @@ -1283,7 +1283,7 @@ mod tests { fn max_l0_overlap_disjoint_ranges_is_one() { // Simulates a post-union manifest where each source's L0s cover // disjoint key ranges — the peak per-point count stays at 1. - let mut l0 = std::collections::VecDeque::new(); + let mut l0 = VecDeque::new(); l0.push_back(create_compacted_sst_view_with_bounds(b"a", Some(b"b"))); l0.push_back(create_compacted_sst_view_with_bounds(b"c", Some(b"d"))); l0.push_back(create_compacted_sst_view_with_bounds(b"e", Some(b"f"))); @@ -1293,7 +1293,7 @@ mod tests { #[test] fn max_l0_overlap_full_overlap_counts_all() { - let mut l0 = std::collections::VecDeque::new(); + let mut l0 = VecDeque::new(); for _ in 0..4 { l0.push_back(create_compacted_sst_view_with_bounds(b"a", Some(b"z"))); } @@ -1303,7 +1303,7 @@ mod tests { #[test] fn max_l0_overlap_partial_overlap() { // A: [a, c], B: [b, d]. At B.start=b, both A and B contain b. - let mut l0 = std::collections::VecDeque::new(); + let mut l0 = VecDeque::new(); l0.push_back(create_compacted_sst_view_with_bounds(b"a", Some(b"c"))); l0.push_back(create_compacted_sst_view_with_bounds(b"b", Some(b"d"))); assert_eq!(super::max_l0_overlap(&l0), 2); @@ -1312,7 +1312,7 @@ mod tests { #[test] fn max_l0_overlap_mixed_disjoint_groups() { // Two disjoint groups of 3 overlapping SSTs each. Peak is 3, not 6. - let mut l0 = std::collections::VecDeque::new(); + let mut l0 = VecDeque::new(); for _ in 0..3 { l0.push_back(create_compacted_sst_view_with_bounds(b"a", Some(b"c"))); } @@ -1325,7 +1325,7 @@ mod tests { #[test] fn max_l0_overlap_single_point_range_is_one() { // A view whose first_entry == last_entry covers exactly one key. - let mut l0 = std::collections::VecDeque::new(); + let mut l0 = VecDeque::new(); l0.push_back(create_compacted_sst_view_with_bounds(b"k", Some(b"k"))); assert_eq!(super::max_l0_overlap(&l0), 1); } @@ -1333,7 +1333,7 @@ mod tests { #[test] fn max_l0_overlap_many_point_ranges_same_key() { // N coincident point ranges [k, k] all cover key k → peak N. - let mut l0 = std::collections::VecDeque::new(); + let mut l0 = VecDeque::new(); for _ in 0..5 { l0.push_back(create_compacted_sst_view_with_bounds(b"k", Some(b"k"))); } @@ -1344,7 +1344,7 @@ mod tests { fn max_l0_overlap_mixed_point_and_longer_ranges_at_same_key() { // Two point ranges [k, k] and two longer ranges [k, z] all cover k. // Peak at k is 4; past k, only the two longer ranges remain. - let mut l0 = std::collections::VecDeque::new(); + let mut l0 = VecDeque::new(); l0.push_back(create_compacted_sst_view_with_bounds(b"k", Some(b"k"))); l0.push_back(create_compacted_sst_view_with_bounds(b"k", Some(b"k"))); l0.push_back(create_compacted_sst_view_with_bounds(b"k", Some(b"z"))); @@ -1355,7 +1355,7 @@ mod tests { #[test] fn max_l0_overlap_edge_touching_inclusive_counts_both() { // [a, b] and [b, c]: both contain b → peak 2. - let mut l0 = std::collections::VecDeque::new(); + let mut l0 = VecDeque::new(); l0.push_back(create_compacted_sst_view_with_bounds(b"a", Some(b"b"))); l0.push_back(create_compacted_sst_view_with_bounds(b"b", Some(b"c"))); assert_eq!(super::max_l0_overlap(&l0), 2); @@ -1367,14 +1367,10 @@ mod tests { // First view has an Excluded end at b via a visible_range projection. let a = Bytes::copy_from_slice(b"a"); let b = Bytes::copy_from_slice(b"b"); - let v1 = create_compacted_sst_view_with_bounds(b"a", Some(b"b")).with_visible_range( - BytesRange::new( - std::ops::Bound::Included(a), - std::ops::Bound::Excluded(b.clone()), - ), - ); + let v1 = create_compacted_sst_view_with_bounds(b"a", Some(b"b")) + .with_visible_range(BytesRange::new(Included(a), Excluded(b.clone()))); let v2 = create_compacted_sst_view_with_bounds(b"b", Some(b"c")); - let mut l0 = std::collections::VecDeque::new(); + let mut l0 = VecDeque::new(); l0.push_back(v1); l0.push_back(v2); assert_eq!(super::max_l0_overlap(&l0), 1); @@ -1384,7 +1380,7 @@ mod tests { fn max_l0_overlap_unbounded_end_single_view() { // A view with first_entry but no last_entry has effective_range // [first, Unbounded) — still one view, peak 1. - let mut l0 = std::collections::VecDeque::new(); + let mut l0 = VecDeque::new(); l0.push_back(create_compacted_sst_view_with_bounds(b"a", None)); assert_eq!(super::max_l0_overlap(&l0), 1); } @@ -1393,7 +1389,7 @@ mod tests { fn max_l0_overlap_unbounded_ends_share_tail() { // [a, ∞) and [b, ∞) both extend to +∞, so they overlap at every // point ≥ b. Peak is 2. - let mut l0 = std::collections::VecDeque::new(); + let mut l0 = VecDeque::new(); l0.push_back(create_compacted_sst_view_with_bounds(b"a", None)); l0.push_back(create_compacted_sst_view_with_bounds(b"b", None)); assert_eq!(super::max_l0_overlap(&l0), 2); @@ -1403,7 +1399,7 @@ mod tests { fn max_l0_overlap_mixed_bounded_and_unbounded_end() { // [a, m] ends at m; [b, ∞) starts before m and extends past it. // They coexist on [b, m]. Peak is 2. - let mut l0 = std::collections::VecDeque::new(); + let mut l0 = VecDeque::new(); l0.push_back(create_compacted_sst_view_with_bounds(b"a", Some(b"m"))); l0.push_back(create_compacted_sst_view_with_bounds(b"b", None)); assert_eq!(super::max_l0_overlap(&l0), 2); @@ -1415,11 +1411,10 @@ mod tests { // Effective range becomes [m, z] (physical end clamps the Unbounded). // Pair with [n, ∞): overlap on [n, z]. Peak is 2. let m = Bytes::copy_from_slice(b"m"); - let projected = create_compacted_sst_view_with_bounds(b"a", Some(b"z")).with_visible_range( - BytesRange::new(std::ops::Bound::Included(m), std::ops::Bound::Unbounded), - ); + let projected = create_compacted_sst_view_with_bounds(b"a", Some(b"z")) + .with_visible_range(BytesRange::new(Included(m), Unbounded)); let open = create_compacted_sst_view_with_bounds(b"n", None); - let mut l0 = std::collections::VecDeque::new(); + let mut l0 = VecDeque::new(); l0.push_back(projected); l0.push_back(open); assert_eq!(super::max_l0_overlap(&l0), 2); @@ -1432,19 +1427,11 @@ mod tests { let lo = Bytes::copy_from_slice(b"a"); let mid = Bytes::copy_from_slice(b"m"); let hi = Bytes::copy_from_slice(b"z"); - let v1 = create_compacted_sst_view_with_bounds(b"a", Some(b"z")).with_visible_range( - BytesRange::new( - std::ops::Bound::Included(lo.clone()), - std::ops::Bound::Excluded(mid.clone()), - ), - ); - let v2 = create_compacted_sst_view_with_bounds(b"a", Some(b"z")).with_visible_range( - BytesRange::new( - std::ops::Bound::Included(mid), - std::ops::Bound::Included(hi), - ), - ); - let mut l0 = std::collections::VecDeque::new(); + let v1 = create_compacted_sst_view_with_bounds(b"a", Some(b"z")) + .with_visible_range(BytesRange::new(Included(lo.clone()), Excluded(mid.clone()))); + let v2 = create_compacted_sst_view_with_bounds(b"a", Some(b"z")) + .with_visible_range(BytesRange::new(Included(mid), Included(hi))); + let mut l0 = VecDeque::new(); l0.push_back(v1); l0.push_back(v2); assert_eq!(super::max_l0_overlap(&l0), 1); @@ -1484,7 +1471,7 @@ mod tests { }); proptest!(ProptestConfig::with_cases(256), |(specs in vec(spec, 0..=8))| { - let mut l0 = std::collections::VecDeque::new(); + let mut l0 = VecDeque::new(); for s in &specs { let view = match &s.end { EndKind::Inclusive(end) => { @@ -1497,8 +1484,8 @@ mod tests { &s.start, None, ) .with_visible_range(BytesRange::new( - std::ops::Bound::Included(s.start.clone()), - std::ops::Bound::Excluded(end.clone()), + Included(s.start.clone()), + Excluded(end.clone()), )), }; l0.push_back(view); diff --git a/slatedb/src/db_status.rs b/slatedb/src/db_status.rs index dfc521bce2..e9b7b6f7ba 100644 --- a/slatedb/src/db_status.rs +++ b/slatedb/src/db_status.rs @@ -67,9 +67,9 @@ impl DbStatus { } } -pub(crate) trait ClosedResultWriter: std::fmt::Debug + Send + Sync + 'static { +pub(crate) trait ClosedResultWriter: fmt::Debug + Send + Sync + 'static { fn write_result(&self, result: Result<(), SlateDBError>); - fn result_reader(&self) -> crate::utils::WatchableOnceCellReader>; + fn result_reader(&self) -> WatchableOnceCellReader>; } /// Manages database lifecycle status, including the close result and @@ -246,7 +246,7 @@ impl ClosedResultWriter for WatchableOnceCell> { self.write(result); } - fn result_reader(&self) -> crate::utils::WatchableOnceCellReader> { + fn result_reader(&self) -> WatchableOnceCellReader> { self.reader() } } @@ -262,7 +262,7 @@ impl ClosedResultWriter for DbStatusManager { } } - fn result_reader(&self) -> crate::utils::WatchableOnceCellReader> { + fn result_reader(&self) -> WatchableOnceCellReader> { self.cell.reader() } } diff --git a/slatedb/src/dispatcher.rs b/slatedb/src/dispatcher.rs index 0dc7aac67d..c6dad6d6aa 100644 --- a/slatedb/src/dispatcher.rs +++ b/slatedb/src/dispatcher.rs @@ -983,7 +983,7 @@ mod test { async fn cleanup( &mut self, - mut messages: futures::stream::BoxStream<'async_trait, TestMessage>, + mut messages: BoxStream<'async_trait, TestMessage>, result: Result<(), SlateDBError>, ) -> Result<(), SlateDBError> { self.cleanup_called.write(result); diff --git a/slatedb/src/error.rs b/slatedb/src/error.rs index 006b6a7a0c..c9bedaa5c2 100644 --- a/slatedb/src/error.rs +++ b/slatedb/src/error.rs @@ -714,8 +714,8 @@ impl From for Error { SlateDBError::CheckpointMissing(_) => Error::data(msg), SlateDBError::InvalidVersion { .. } => Error::data(msg), SlateDBError::ManifestMissing(_) => Error::data(msg), - SlateDBError::LatestTransactionalObjectVersionMissing => Error::data(msg), - SlateDBError::TransactionalObjectVersionExists => Error::data(msg), + LatestTransactionalObjectVersionMissing => Error::data(msg), + TransactionalObjectVersionExists => Error::data(msg), SlateDBError::InvalidTransactionalObjectState => Error::data(msg), SlateDBError::EmptyManifest => Error::data(msg), SlateDBError::EmptyBlock => Error::data(msg), diff --git a/slatedb/src/filter.rs b/slatedb/src/filter.rs index 8239e22503..453c42087d 100644 --- a/slatedb/src/filter.rs +++ b/slatedb/src/filter.rs @@ -113,7 +113,7 @@ impl BloomFilter { /// checksum, which are accounted for at the SST level. pub(crate) fn estimate_encoded_size(num_keys: u32, filter_bits_per_key: u32) -> usize { let filter_bytes = BloomFilterBuilder::filter_size_bytes(num_keys, filter_bits_per_key); - let num_probes_size = std::mem::size_of::(); + let num_probes_size = size_of::(); filter_bytes + num_probes_size } diff --git a/slatedb/src/flatbuffer_types.rs b/slatedb/src/flatbuffer_types.rs index e23347239b..1bf6bd2023 100644 --- a/slatedb/src/flatbuffer_types.rs +++ b/slatedb/src/flatbuffer_types.rs @@ -464,7 +464,7 @@ impl FlatBufferManifestCodec { } fn decode_sorted_runs_v2( - runs: flatbuffers::Vector<'_, flatbuffers::ForwardsUOffset>>, + runs: Vector<'_, ForwardsUOffset>>, sst_lookup: &std::collections::HashMap, ) -> Result, Box> { runs.iter() @@ -503,8 +503,8 @@ impl FlatBufferManifestCodec { /// and a sorted-runs vector. fn decode_lsm_tree_v2( last_compacted_l0_sst_view_id: Option, - l0: flatbuffers::Vector<'_, flatbuffers::ForwardsUOffset>>, - compacted: flatbuffers::Vector<'_, flatbuffers::ForwardsUOffset>>, + l0: Vector<'_, ForwardsUOffset>>, + compacted: Vector<'_, ForwardsUOffset>>, sst_lookup: &std::collections::HashMap, ) -> Result> { let last_compacted_l0_sst_view_id = last_compacted_l0_sst_view_id.map(|id| id.ulid()); @@ -837,7 +837,7 @@ impl<'b> DbFlatBufferBuilder<'b> { SsTableId::Wal(_) => { unreachable!("cannot pass WAL SST handle to create compacted sst") } - SsTableId::Compacted(ulid) => ulid, + Compacted(ulid) => ulid, }; let compacted_sst_id = self.add_compacted_sst_id(&ulid); let compacted_sst_info = self.add_sst_info(&handle.info); @@ -872,7 +872,7 @@ impl<'b> DbFlatBufferBuilder<'b> { SsTableId::Wal(_) => { unreachable!("cannot pass WAL SST handle to create compacted sst from view") } - SsTableId::Compacted(ulid) => ulid, + Compacted(ulid) => ulid, }; let compacted_sst_id = self.add_compacted_sst_id(&ulid); let compacted_sst_info = self.add_sst_info(&view.sst.info); @@ -896,7 +896,7 @@ impl<'b> DbFlatBufferBuilder<'b> { SsTableId::Wal(_) => { unreachable!("cannot pass WAL SST handle to create compacted sst v2") } - SsTableId::Compacted(ulid) => ulid, + Compacted(ulid) => ulid, }; let compacted_sst_id = self.add_compacted_sst_id(&ulid); let compacted_sst_info = self.add_sst_info(&handle.info); @@ -918,7 +918,7 @@ impl<'b> DbFlatBufferBuilder<'b> { SsTableId::Wal(_) => { unreachable!("cannot pass WAL SST handle to create compacted sst view") } - SsTableId::Compacted(ulid) => ulid, + Compacted(ulid) => ulid, }; let sst_id = self.add_compacted_sst_id(&ulid); let visible_range = view.visible_range.as_ref().map(|r| self.add_bytes_range(r)); @@ -1279,13 +1279,13 @@ impl<'b> DbFlatBufferBuilder<'b> { std::collections::BTreeMap::new(); for tree in core.trees() { for view in tree.l0.iter() { - if let SsTableId::Compacted(ulid) = view.sst.id { + if let Compacted(ulid) = view.sst.id { unique_ssts.entry(ulid).or_insert(&view.sst); } } for sr in tree.compacted.iter() { for view in sr.sst_views() { - if let SsTableId::Compacted(ulid) = view.sst.id { + if let Compacted(ulid) = view.sst.id { unique_ssts.entry(ulid).or_insert(&view.sst); } } @@ -2155,7 +2155,7 @@ mod tests { ]; for status in statuses { - let fb_status = super::FbCompactionStatus::from(status); + let fb_status = FbCompactionStatus::from(status); let round_trip = CompactionStatus::from(fb_status); assert_eq!(round_trip, status); } @@ -2167,7 +2167,7 @@ mod tests { let compactions = Compactions::new(1); let bytes = codec.encode(&compactions); - let invalid_version = super::COMPACTIONS_FORMAT_VERSION + 1; + let invalid_version = COMPACTIONS_FORMAT_VERSION + 1; let mut invalid_bytes = bytes.to_vec(); invalid_bytes[0] = (invalid_version >> 8) as u8; invalid_bytes[1] = invalid_version as u8; @@ -2359,9 +2359,9 @@ mod tests { }, ); let l0_ulid = ulid::Ulid::new(); - let l0_id = super::root_generated::Ulid::create( + let l0_id = root_generated::Ulid::create( &mut fbb, - &super::root_generated::UlidArgs { + &root_generated::UlidArgs { high: (l0_ulid.0 >> 64) as u64, low: ((l0_ulid.0 << 64) >> 64) as u64, }, @@ -2386,9 +2386,9 @@ mod tests { }, ); let sr_ulid = ulid::Ulid::new(); - let sr_id = super::root_generated::Ulid::create( + let sr_id = root_generated::Ulid::create( &mut fbb, - &super::root_generated::UlidArgs { + &root_generated::UlidArgs { high: (sr_ulid.0 >> 64) as u64, low: ((sr_ulid.0 << 64) >> 64) as u64, }, @@ -2411,8 +2411,8 @@ mod tests { }, ); let compacted_vec = fbb.create_vector(&[sorted_run]); - let checkpoints_vec = fbb - .create_vector::>(&[]); + let checkpoints_vec = + fbb.create_vector::>(&[]); let manifest = ManifestV1::create( &mut fbb, &ManifestV1Args { @@ -2509,9 +2509,9 @@ mod tests { }, ); let sst_ulid = ulid::Ulid::new(); - let sst_id = super::root_generated::Ulid::create( + let sst_id = root_generated::Ulid::create( &mut fbb, - &super::root_generated::UlidArgs { + &root_generated::UlidArgs { high: (sst_ulid.0 >> 64) as u64, low: ((sst_ulid.0 << 64) >> 64) as u64, }, @@ -2528,9 +2528,9 @@ mod tests { let output_ssts_vec = fbb.create_vector(&[output_sst]); // Build a compaction with a tiered spec let source_ulid = ulid::Ulid::new(); - let source_id = super::root_generated::Ulid::create( + let source_id = root_generated::Ulid::create( &mut fbb, - &super::root_generated::UlidArgs { + &root_generated::UlidArgs { high: (source_ulid.0 >> 64) as u64, low: ((source_ulid.0 << 64) >> 64) as u64, }, @@ -2548,9 +2548,9 @@ mod tests { }, ); let compaction_ulid = ulid::Ulid::new(); - let compaction_id = super::root_generated::Ulid::create( + let compaction_id = root_generated::Ulid::create( &mut fbb, - &super::root_generated::UlidArgs { + &root_generated::UlidArgs { high: (compaction_ulid.0 >> 64) as u64, low: ((compaction_ulid.0 << 64) >> 64) as u64, }, @@ -2564,7 +2564,7 @@ mod tests { status: fb_status, output_ssts: Some(output_ssts_vec), worker: None, - ctx_type: super::root_generated::CompactionContext::NONE, + ctx_type: root_generated::CompactionContext::NONE, ctx: None, }, ); @@ -2605,9 +2605,9 @@ mod tests { }; let mut fbb = flatbuffers::FlatBufferBuilder::new(); let source_ulid = ulid::Ulid::new(); - let source_id = super::root_generated::Ulid::create( + let source_id = root_generated::Ulid::create( &mut fbb, - &super::root_generated::UlidArgs { + &root_generated::UlidArgs { high: (source_ulid.0 >> 64) as u64, low: ((source_ulid.0 << 64) >> 64) as u64, }, @@ -2624,9 +2624,9 @@ mod tests { }, ); let compaction_ulid = ulid::Ulid::new(); - let compaction_id = super::root_generated::Ulid::create( + let compaction_id = root_generated::Ulid::create( &mut fbb, - &super::root_generated::UlidArgs { + &root_generated::UlidArgs { high: (compaction_ulid.0 >> 64) as u64, low: ((compaction_ulid.0 << 64) >> 64) as u64, }, @@ -2640,7 +2640,7 @@ mod tests { status: FbCompactionStatus::Running, output_ssts: None, worker: None, - ctx_type: super::root_generated::CompactionContext::NONE, + ctx_type: root_generated::CompactionContext::NONE, ctx: None, }, ); @@ -2692,9 +2692,9 @@ mod tests { }, ); let compaction_ulid = ulid::Ulid::new(); - let compaction_id = super::root_generated::Ulid::create( + let compaction_id = root_generated::Ulid::create( &mut fbb, - &super::root_generated::UlidArgs { + &root_generated::UlidArgs { high: (compaction_ulid.0 >> 64) as u64, low: ((compaction_ulid.0 << 64) >> 64) as u64, }, @@ -2708,7 +2708,7 @@ mod tests { status: FbCompactionStatus::Running, output_ssts: None, worker: None, - ctx_type: super::root_generated::CompactionContext::NONE, + ctx_type: root_generated::CompactionContext::NONE, ctx: None, }, ); diff --git a/slatedb/src/format/block.rs b/slatedb/src/format/block.rs index 02b1c79f1b..f9b22a1f61 100644 --- a/slatedb/src/format/block.rs +++ b/slatedb/src/format/block.rs @@ -5,7 +5,7 @@ use crate::types::RowEntry; use crate::utils::clamp_allocated_size_bytes; use bytes::{Buf, BufMut, Bytes, BytesMut}; -pub(crate) const SIZEOF_U16: usize = std::mem::size_of::(); +pub(crate) const SIZEOF_U16: usize = size_of::(); #[derive(Eq, PartialEq)] pub(crate) struct Block { diff --git a/slatedb/src/format/row.rs b/slatedb/src/format/row.rs index 10e6905710..41c7583445 100644 --- a/slatedb/src/format/row.rs +++ b/slatedb/src/format/row.rs @@ -147,10 +147,10 @@ impl SstRowCodecV0 { /// estimated_entries_size include the size of seqnum,create_ts(if exist),expire_ts(exist),key,value pub(crate) fn estimate_encoded_size(entry_num: usize, estimated_entries_size: usize) -> usize { - let key_prefix_len_size = std::mem::size_of::(); - let key_suffix_len_size = std::mem::size_of::(); - let value_len_size = std::mem::size_of::(); - let flag_size = std::mem::size_of::(); + let key_prefix_len_size = size_of::(); + let key_suffix_len_size = size_of::(); + let value_len_size = size_of::(); + let flag_size = size_of::(); let mut ans = estimated_entries_size; ans += (key_prefix_len_size + key_suffix_len_size + value_len_size + flag_size) * entry_num; ans diff --git a/slatedb/src/format/sst.rs b/slatedb/src/format/sst.rs index 7a34982efd..7413c98615 100644 --- a/slatedb/src/format/sst.rs +++ b/slatedb/src/format/sst.rs @@ -69,10 +69,7 @@ impl BlockBuilder { } } - pub(crate) fn add( - &mut self, - entry: crate::types::RowEntry, - ) -> Result { + pub(crate) fn add(&mut self, entry: crate::types::RowEntry) -> Result { match self { Self::V1(builder) => builder.add(entry), Self::V2(builder) => builder.add(entry), @@ -125,10 +122,7 @@ impl BlockBuilderWithStats { self.builder.would_fit(entry) } - pub(crate) fn add( - &mut self, - entry: crate::types::RowEntry, - ) -> Result { + pub(crate) fn add(&mut self, entry: crate::types::RowEntry) -> Result { match &entry.value { crate::types::ValueDeletable::Value(_) => self.stats.num_puts += 1, crate::types::ValueDeletable::Merge(_) => self.stats.num_merges += 1, @@ -329,7 +323,7 @@ pub(crate) struct EncodedSsTableFooterBuilder<'a, 'b> { /// codec for the SST info sst_info_codec: &'a dyn SsTableInfoCodec, /// builder for the index block - index_builder: flatbuffers::FlatBufferBuilder<'b, flatbuffers::DefaultAllocator>, + index_builder: flatbuffers::FlatBufferBuilder<'b, DefaultAllocator>, /// metadata block block_meta: Vec>>, /// filter blocks diff --git a/slatedb/src/garbage_collector.rs b/slatedb/src/garbage_collector.rs index ad9fe783ae..674840ad39 100644 --- a/slatedb/src/garbage_collector.rs +++ b/slatedb/src/garbage_collector.rs @@ -710,7 +710,7 @@ mod tests { fn new_checkpoint(manifest_id: u64, expire_time: Option>) -> Checkpoint { Checkpoint { - id: uuid::Uuid::new_v4(), + id: Uuid::new_v4(), manifest_id, expire_time, create_time: DefaultSystemClock::default().now(), @@ -1283,7 +1283,7 @@ mod tests { assert_eq!( lookup_metric_with_labels( &recorder, - crate::garbage_collector::stats::DELETED_COUNT, + stats::DELETED_COUNT, &[("resource", "wal_fence")] ), Some(2) @@ -1848,23 +1848,23 @@ mod tests { let gc_opts = GarbageCollectorOptions { manifest_options: Some(GarbageCollectorDirectoryOptions { - min_age: std::time::Duration::from_secs(3600), + min_age: Duration::from_secs(3600), interval: None, dry_run: false, }), - wal_options: Some(crate::config::GarbageCollectorDirectoryOptions { - min_age: std::time::Duration::from_secs(3600), + wal_options: Some(GarbageCollectorDirectoryOptions { + min_age: Duration::from_secs(3600), interval: None, dry_run: false, }), wal_fence_options: None, - compacted_options: Some(crate::config::GarbageCollectorDirectoryOptions { - min_age: std::time::Duration::from_secs(3600), + compacted_options: Some(GarbageCollectorDirectoryOptions { + min_age: Duration::from_secs(3600), interval: None, dry_run: false, }), - compactions_options: Some(crate::config::GarbageCollectorDirectoryOptions { - min_age: std::time::Duration::from_secs(3600), + compactions_options: Some(GarbageCollectorDirectoryOptions { + min_age: Duration::from_secs(3600), interval: None, dry_run: false, }), @@ -1926,23 +1926,23 @@ mod tests { let recorder = MetricsRecorderHelper::noop(); let gc_opts = GarbageCollectorOptions { manifest_options: Some(GarbageCollectorDirectoryOptions { - min_age: std::time::Duration::from_secs(3600), + min_age: Duration::from_secs(3600), interval: None, dry_run: false, }), wal_options: Some(GarbageCollectorDirectoryOptions { - min_age: std::time::Duration::from_secs(3600), + min_age: Duration::from_secs(3600), interval: None, dry_run: false, }), wal_fence_options: None, compacted_options: Some(GarbageCollectorDirectoryOptions { - min_age: std::time::Duration::from_secs(3600), + min_age: Duration::from_secs(3600), interval: None, dry_run: false, }), compactions_options: Some(GarbageCollectorDirectoryOptions { - min_age: std::time::Duration::from_secs(3600), + min_age: Duration::from_secs(3600), interval: None, dry_run: false, }), @@ -2004,18 +2004,18 @@ mod tests { let gc_opts = GarbageCollectorOptions { manifest_options: None, wal_options: Some(GarbageCollectorDirectoryOptions { - min_age: std::time::Duration::from_secs(3600), + min_age: Duration::from_secs(3600), interval: None, dry_run: false, }), wal_fence_options: None, compacted_options: Some(GarbageCollectorDirectoryOptions { - min_age: std::time::Duration::from_secs(3600), + min_age: Duration::from_secs(3600), interval: None, dry_run: false, }), compactions_options: Some(GarbageCollectorDirectoryOptions { - min_age: std::time::Duration::from_secs(3600), + min_age: Duration::from_secs(3600), interval: None, dry_run: false, }), @@ -2111,18 +2111,18 @@ mod tests { interval: Some(Duration::from_secs(1)), dry_run: false, }), - wal_options: Some(crate::config::GarbageCollectorDirectoryOptions { + wal_options: Some(GarbageCollectorDirectoryOptions { min_age: Duration::from_secs(3600), interval: Some(Duration::from_secs(1)), dry_run: false, }), wal_fence_options: None, - compacted_options: Some(crate::config::GarbageCollectorDirectoryOptions { + compacted_options: Some(GarbageCollectorDirectoryOptions { min_age: Duration::from_secs(3600), interval: Some(Duration::from_secs(1)), dry_run: false, }), - compactions_options: Some(crate::config::GarbageCollectorDirectoryOptions { + compactions_options: Some(GarbageCollectorDirectoryOptions { min_age: Duration::from_secs(3600), interval: Some(Duration::from_secs(1)), dry_run: false, @@ -2184,11 +2184,7 @@ mod tests { // then: assert_eq!( - lookup_metric_with_labels( - &recorder, - crate::garbage_collector::stats::DELETED_COUNT, - &[("resource", "manifest")] - ), + lookup_metric_with_labels(&recorder, stats::DELETED_COUNT, &[("resource", "manifest")]), Some(1) ); } @@ -2229,11 +2225,7 @@ mod tests { // then: assert_eq!( - lookup_metric_with_labels( - &recorder, - crate::garbage_collector::stats::DELETED_COUNT, - &[("resource", "wal")] - ), + lookup_metric_with_labels(&recorder, stats::DELETED_COUNT, &[("resource", "wal")]), Some(1) ); } @@ -2290,7 +2282,7 @@ mod tests { assert_eq!( lookup_metric_with_labels( &recorder, - crate::garbage_collector::stats::DELETED_COUNT, + stats::DELETED_COUNT, &[("resource", "compacted")] ), Some(1) @@ -2354,7 +2346,7 @@ mod tests { assert_eq!( lookup_metric_with_labels( &recorder, - crate::garbage_collector::stats::DELETED_COUNT, + stats::DELETED_COUNT, &[("resource", "compactions")] ), Some(2) @@ -2594,11 +2586,7 @@ mod tests { .collect::>(); assert_eq!(wal_ids, vec![rejected_before_wal_id, rejected_after_wal_id]); assert_eq!( - lookup_metric_with_labels( - &recorder, - crate::garbage_collector::stats::DELETED_COUNT, - &[("resource", "wal")] - ), + lookup_metric_with_labels(&recorder, stats::DELETED_COUNT, &[("resource", "wal")]), Some(1) ); } diff --git a/slatedb/src/lib.rs b/slatedb/src/lib.rs index 92131c15d3..3bfd81214e 100644 --- a/slatedb/src/lib.rs +++ b/slatedb/src/lib.rs @@ -184,5 +184,5 @@ mod wal_replay; #[cfg(test)] #[ctor::ctor] fn init_test_infrastructure() { - crate::test_utils::init_test_infrastructure(); + test_utils::init_test_infrastructure(); } diff --git a/slatedb/src/manifest/mod.rs b/slatedb/src/manifest/mod.rs index bd4644f6b1..f8af20347e 100644 --- a/slatedb/src/manifest/mod.rs +++ b/slatedb/src/manifest/mod.rs @@ -1599,7 +1599,7 @@ mod tests { .await .unwrap(); let checkpoint = parent_manifest - .write_checkpoint(uuid::Uuid::new_v4(), &CheckpointOptions::default()) + .write_checkpoint(Uuid::new_v4(), &CheckpointOptions::default()) .await .unwrap(); @@ -1741,7 +1741,7 @@ mod tests { .unwrap(); let checkpoint = manifest - .write_checkpoint(uuid::Uuid::new_v4(), &CheckpointOptions::default()) + .write_checkpoint(Uuid::new_v4(), &CheckpointOptions::default()) .await .unwrap(); @@ -1990,7 +1990,7 @@ mod tests { fn test_union(#[case] test_case: UnionTestCase) { let mut sst_ids: HashMap = HashMap::new(); let rand = Arc::new(DbRand::default()); - let sources: Vec = test_case + let sources: Vec = test_case .manifests .iter() .enumerate() @@ -2755,7 +2755,7 @@ mod tests { ); // Verify no duplicates - let mut seen = std::collections::HashSet::new(); + let mut seen = HashSet::new(); for id in &sr_ids { assert!(seen.insert(id), "Duplicate SR ID: {}", id); } @@ -3702,14 +3702,14 @@ mod tests { ); } - fn segment_with_prefix(prefix: &[u8], seed: u64) -> super::Segment { + fn segment_with_prefix(prefix: &[u8], seed: u64) -> Segment { let view_id = Ulid::from_parts(seed, 0); let handle = SsTableHandle::new( SsTableId::Compacted(Ulid::from_parts(seed, 1)), SST_FORMAT_VERSION_LATEST, SsTableInfo::default(), ); - super::Segment { + Segment { prefix: Bytes::copy_from_slice(prefix), tree: Arc::new(LsmTreeState { last_compacted_l0_sst_view_id: None, @@ -3720,7 +3720,7 @@ mod tests { } } - fn collect_prefixes(segments: &[super::Segment]) -> Vec { + fn collect_prefixes(segments: &[Segment]) -> Vec { segments.iter().map(|s| s.prefix.clone()).collect() } diff --git a/slatedb/src/manifest/store.rs b/slatedb/src/manifest/store.rs index 8707dc04b0..79a491a547 100644 --- a/slatedb/src/manifest/store.rs +++ b/slatedb/src/manifest/store.rs @@ -584,7 +584,6 @@ pub(crate) mod test_utils { mod tests { use crate::checkpoint::Checkpoint; use crate::config::CheckpointOptions; - use crate::error; use crate::error::SlateDBError; use crate::manifest::store::{FenceableManifest, ManifestStore, StoredManifest}; use crate::manifest::ManifestCore; @@ -622,7 +621,7 @@ mod tests { assert!(matches!( result.unwrap_err(), - error::SlateDBError::TransactionalObjectVersionExists + SlateDBError::TransactionalObjectVersionExists )); } @@ -755,7 +754,7 @@ mod tests { .unwrap(); let result = writer1.refresh().await; - assert!(matches!(result, Err(error::SlateDBError::Fenced))); + assert!(matches!(result, Err(SlateDBError::Fenced))); } #[tokio::test] @@ -807,7 +806,7 @@ mod tests { .unwrap(); let result = compactor1.refresh().await; - assert!(matches!(result, Err(error::SlateDBError::Fenced))); + assert!(matches!(result, Err(SlateDBError::Fenced))); } #[tokio::test] @@ -859,7 +858,7 @@ mod tests { .write_checkpoint(uuid::Uuid::new_v4(), &CheckpointOptions::default()) .await; - assert!(matches!(result, Err(error::SlateDBError::Fenced))); + assert!(matches!(result, Err(SlateDBError::Fenced))); assert_state_not_updated(&mut compactor2).await; } @@ -1052,7 +1051,7 @@ mod tests { assert_eq!(manifests[1].id, 2); let result = ms.delete_manifest(2).await; - assert!(matches!(result, Err(error::SlateDBError::InvalidDeletion))); + assert!(matches!(result, Err(SlateDBError::InvalidDeletion))); } fn new_memory_manifest_store() -> Arc { diff --git a/slatedb/src/mem_table.rs b/slatedb/src/mem_table.rs index 21155b2f1c..f29a1d17b2 100644 --- a/slatedb/src/mem_table.rs +++ b/slatedb/src/mem_table.rs @@ -529,13 +529,12 @@ impl KVTable { // because the monotonicity is enforced when generating the clock tick // (see [crate::utils::MonotonicClock::now]) if let Some(create_ts) = row.create_ts { - self.last_tick - .fetch_max(create_ts, atomic::Ordering::SeqCst); + self.last_tick.fetch_max(create_ts, SeqCst); } // update the last seq number if it is greater than the current last seq - self.last_seq.fetch_max(row.seq, atomic::Ordering::SeqCst); + self.last_seq.fetch_max(row.seq, SeqCst); // update the first seq number if it is smaller than the current first seq - self.first_seq.fetch_min(row.seq, atomic::Ordering::SeqCst); + self.first_seq.fetch_min(row.seq, SeqCst); let row_size = row.estimated_size(); self.map.compare_insert(internal_key, row, |previous_row| { diff --git a/slatedb/src/memtable_flusher/manifest_writer.rs b/slatedb/src/memtable_flusher/manifest_writer.rs index babc6bbdd2..a58aebc255 100644 --- a/slatedb/src/memtable_flusher/manifest_writer.rs +++ b/slatedb/src/memtable_flusher/manifest_writer.rs @@ -1365,7 +1365,7 @@ mod tests { Duration::from_secs(3600), ); - let (tx, rx) = tokio::sync::oneshot::channel(); + let (tx, rx) = oneshot::channel(); started .send_checkpoint(None, CheckpointOptions::default(), tx) .unwrap(); @@ -1822,7 +1822,7 @@ mod tests { // manifest writer routes by `prefix` from the surrounding // `SegmentedSstHandle`, not by the SST's keys. let mut builder = inner.table_store.table_builder(); - let row = crate::types::RowEntry::new_value(prefix, value, first_seq); + let row = RowEntry::new_value(prefix, value, first_seq); builder.add(row).await.unwrap(); let encoded_sst = builder.build().await.unwrap(); let id = crate::db_state::SsTableId::Compacted( diff --git a/slatedb/src/memtable_flusher/mod.rs b/slatedb/src/memtable_flusher/mod.rs index c8602db624..c2221afb01 100644 --- a/slatedb/src/memtable_flusher/mod.rs +++ b/slatedb/src/memtable_flusher/mod.rs @@ -47,8 +47,8 @@ pub(crate) enum FlushTarget { /// Parallel L0 memtable flusher subsystem. pub(crate) struct MemtableFlusher { - messages_tx: SafeSender, - messages_rx: async_channel::Receiver, + messages_tx: SafeSender, + messages_rx: async_channel::Receiver, } impl MemtableFlusher { @@ -112,15 +112,14 @@ impl MemtableFlusher { pub(crate) async fn flush(&self, target: FlushTarget) -> Result { let (tx, rx) = oneshot::channel(); self.messages_tx - .send(tracker::TrackerMessage::FlushRequest { target, sender: tx })?; + .send(TrackerMessage::FlushRequest { target, sender: tx })?; rx.await.map_err(SlateDBError::ReadChannelError)? } /// Notifies the flusher that a memtable may have been frozen. /// Triggers reconcile and dispatch without waiting for a result. pub(crate) fn notify_memtable_frozen(&self) -> Result<(), SlateDBError> { - self.messages_tx - .send(tracker::TrackerMessage::MemtableFrozen) + self.messages_tx.send(TrackerMessage::MemtableFrozen) } /// Creates a checkpoint using the memtable flusher's flush semantics. @@ -130,12 +129,11 @@ impl MemtableFlusher { options: CheckpointOptions, ) -> Result { let (tx, rx) = oneshot::channel(); - self.messages_tx - .send(tracker::TrackerMessage::CheckpointRequest { - target, - options, - sender: tx, - })?; + self.messages_tx.send(TrackerMessage::CheckpointRequest { + target, + options, + sender: tx, + })?; rx.await.map_err(SlateDBError::ReadChannelError)? } diff --git a/slatedb/src/memtable_flusher/tracker.rs b/slatedb/src/memtable_flusher/tracker.rs index f608b22386..ebbff49056 100644 --- a/slatedb/src/memtable_flusher/tracker.rs +++ b/slatedb/src/memtable_flusher/tracker.rs @@ -271,7 +271,7 @@ impl FlushTracker { /// When the imm's touched-segment set is empty (no extractor /// configured, or the imm came from a path that bypassed /// validation) we fall back to the max-across-trees heuristic. - fn can_dispatch(&self, imm: &crate::mem_table::ImmutableMemtable) -> bool { + fn can_dispatch(&self, imm: &ImmutableMemtable) -> bool { let state = self.inner.state.read().state(); let core = state.core(); let settings = &self.inner.settings; @@ -413,7 +413,7 @@ fn allocate_segment_sst_ids(inner: &DbInner, imm: &ImmutableMemtable) -> BTreeMa struct TrackedImm { first_seq: u64, last_seq: u64, - imm_memtable: Arc, + imm_memtable: Arc, state: TrackedImmState, } @@ -432,10 +432,7 @@ impl TrackedImmFrontier { } /// Register newly frozen immutable memtables, deduplicating by `last_seq`. - fn register( - &mut self, - imm_memtables: impl Iterator>, - ) { + fn register(&mut self, imm_memtables: impl Iterator>) { for imm_memtable in imm_memtables { let first_seq = imm_memtable .table() diff --git a/slatedb/src/memtable_flusher/uploader.rs b/slatedb/src/memtable_flusher/uploader.rs index 271466dcd0..9865975193 100644 --- a/slatedb/src/memtable_flusher/uploader.rs +++ b/slatedb/src/memtable_flusher/uploader.rs @@ -426,12 +426,7 @@ mod tests { ) } - fn freeze_imm( - db: &DbInner, - key: &[u8], - value: &[u8], - seq: u64, - ) -> Arc { + fn freeze_imm(db: &DbInner, key: &[u8], value: &[u8], seq: u64) -> Arc { let mut guard = db.state.write(); guard.memtable().put(RowEntry::new_value(key, value, seq)); guard.freeze_memtable(0); @@ -653,11 +648,9 @@ mod tests { .await; { let mut guard = db.state.write(); - guard.memtable().put(crate::types::RowEntry::new_merge( - b"key", - b"merge_operand", - 1, - )); + guard + .memtable() + .put(RowEntry::new_merge(b"key", b"merge_operand", 1)); guard.freeze_memtable(0); } let imm_memtable = db @@ -731,7 +724,7 @@ mod tests { let mut guard = db.state.write(); guard .memtable() - .put(crate::types::RowEntry::new_merge(b"key", b"operand", 1)); + .put(RowEntry::new_merge(b"key", b"operand", 1)); guard.freeze_memtable(0); } let imm_memtable = db diff --git a/slatedb/src/merge_operator.rs b/slatedb/src/merge_operator.rs index 6e21f78e29..2472e57f09 100644 --- a/slatedb/src/merge_operator.rs +++ b/slatedb/src/merge_operator.rs @@ -523,12 +523,12 @@ mod tests { /// Mock merge operator that tracks whether merge_batch is called struct MockBatchedMergeOperator { - merge_batch_call_count: std::sync::Arc, + merge_batch_call_count: Arc, } impl MockBatchedMergeOperator { - fn new() -> (Self, std::sync::Arc) { - let counter = std::sync::Arc::new(std::sync::atomic::AtomicUsize::new(0)); + fn new() -> (Self, Arc) { + let counter = Arc::new(std::sync::atomic::AtomicUsize::new(0)); ( Self { merge_batch_call_count: counter.clone(), diff --git a/slatedb/src/paths.rs b/slatedb/src/paths.rs index a909a9615c..2bd157b9d9 100644 --- a/slatedb/src/paths.rs +++ b/slatedb/src/paths.rs @@ -76,13 +76,13 @@ impl PathResolver { .next() .and_then(|s| s.as_ref().split('.').next().map(|s| s.parse::())) .transpose() - .map(|r| r.map(SsTableId::Wal)) + .map(|r| r.map(Wal)) .map_err(|_| SlateDBError::InvalidDBState), Some(a) if a.as_ref() == COMPACTED_PATH => suffix_iter .next() .and_then(|s| s.as_ref().split('.').next().map(Ulid::from_string)) .transpose() - .map(|r| r.map(SsTableId::Compacted)) + .map(|r| r.map(Compacted)) .map_err(|_| SlateDBError::InvalidDBState), _ => Ok(None), } diff --git a/slatedb/src/reader.rs b/slatedb/src/reader.rs index 1b90816fc5..1187c0c7c5 100644 --- a/slatedb/src/reader.rs +++ b/slatedb/src/reader.rs @@ -1262,7 +1262,7 @@ mod tests { let write_batch = populate_db_state(&mut test_db_state, test_case.entries).await?; // Create Reader with test clock - let recorder = slatedb_common::metrics::MetricsRecorderHelper::noop(); + let recorder = MetricsRecorderHelper::noop(); let db_stats = DbStats::new(&recorder); let test_clock = Arc::new(MockSystemClock::new()); let mono_clock = Arc::new(MonotonicClock::new(test_clock as Arc, 0)); @@ -1692,7 +1692,7 @@ mod tests { let write_batch = populate_db_state(&mut test_db_state, test_case.entries).await?; // Create Reader with test clock - let recorder = slatedb_common::metrics::MetricsRecorderHelper::noop(); + let recorder = MetricsRecorderHelper::noop(); let db_stats = DbStats::new(&recorder); let test_clock = Arc::new(MockSystemClock::new()); let mono_clock = Arc::new(MonotonicClock::new(test_clock as Arc, 0)); @@ -1971,7 +1971,7 @@ mod tests { let mut test_db_state = TestDbState::new().await; let write_batch = populate_db_state(&mut test_db_state, entries).await?; - let recorder = slatedb_common::metrics::MetricsRecorderHelper::noop(); + let recorder = MetricsRecorderHelper::noop(); let db_stats = DbStats::new(&recorder); let reader = build_reader(&test_db_state, db_stats, false).await; @@ -2018,7 +2018,7 @@ mod tests { let mut test_db_state = TestDbState::new().await; let write_batch = populate_db_state(&mut test_db_state, entries).await?; - let recorder = slatedb_common::metrics::MetricsRecorderHelper::noop(); + let recorder = MetricsRecorderHelper::noop(); let db_stats = DbStats::new(&recorder); let reader = build_reader(&test_db_state, db_stats, false).await; @@ -2084,7 +2084,7 @@ mod tests { let mut test_db_state = TestDbState::new().await; let write_batch = populate_db_state(&mut test_db_state, entries).await?; - let recorder = slatedb_common::metrics::MetricsRecorderHelper::noop(); + let recorder = MetricsRecorderHelper::noop(); let db_stats = DbStats::new(&recorder); let reader = build_reader(&test_db_state, db_stats, true).await; @@ -2124,7 +2124,7 @@ mod tests { let mut test_db_state = TestDbState::new().await; let write_batch = populate_db_state(&mut test_db_state, entries).await?; - let recorder = slatedb_common::metrics::MetricsRecorderHelper::noop(); + let recorder = MetricsRecorderHelper::noop(); let db_stats = DbStats::new(&recorder); let reader = build_reader(&test_db_state, db_stats, true).await; diff --git a/slatedb/src/retention_iterator.rs b/slatedb/src/retention_iterator.rs index bdb1d0af8b..b96a48107e 100644 --- a/slatedb/src/retention_iterator.rs +++ b/slatedb/src/retention_iterator.rs @@ -1028,7 +1028,7 @@ mod tests { use slatedb_common::clock::MockSystemClock; // Test the apply_retention_filter function directly since TestIterator doesn't support create_ts - let mut versions = std::collections::BTreeMap::new(); + let mut versions = BTreeMap::new(); for entry in test_case.input_entries.iter() { versions.insert(Reverse(entry.seq), entry.clone()); } @@ -1238,7 +1238,7 @@ mod tests { assert_eq!(filtered.len(), 1); let only = filtered.values().next().unwrap(); assert!( - matches!(only.value, ValueDeletable::Tombstone), + matches!(only.value, Tombstone), "expired value should become a tombstone, got {:?}", only.value ); @@ -1318,7 +1318,7 @@ mod tests { // merge dropped, value kept as tombstone assert_eq!(filtered.len(), 1); let kept = filtered.values().next().unwrap(); - assert!(matches!(kept.value, ValueDeletable::Tombstone)); + assert!(matches!(kept.value, Tombstone)); } } } diff --git a/slatedb/src/retrying_object_store.rs b/slatedb/src/retrying_object_store.rs index b054a04a8d..9a9f71b389 100644 --- a/slatedb/src/retrying_object_store.rs +++ b/slatedb/src/retrying_object_store.rs @@ -592,7 +592,7 @@ mod tests { #[tokio::test] async fn test_put_opts_retries_transient_until_success() { - let inner: Arc = Arc::new(InMemory::new()); + let inner: Arc = Arc::new(InMemory::new()); let flaky = Arc::new(FlakyObjectStore::new(inner, 1)); let retrying = RetryingObjectStore::new(flaky.clone(), test_rand(), test_clock(), None); @@ -663,7 +663,7 @@ mod tests { #[tokio::test] async fn test_put_opts_retry_sleep_uses_system_clock() { - let inner: Arc = Arc::new(InMemory::new()); + let inner: Arc = Arc::new(InMemory::new()); let flaky = Arc::new(FlakyObjectStore::new(inner, 1)); let clock = Arc::new(MockSystemClock::new()); let retrying = RetryingObjectStore::new(flaky.clone(), test_rand(), clock.clone(), None); @@ -707,7 +707,7 @@ mod tests { #[tokio::test] async fn test_put_opts_does_not_retry_on_already_exists() { - let inner: Arc = Arc::new(InMemory::new()); + let inner: Arc = Arc::new(InMemory::new()); let flaky = Arc::new(FlakyObjectStore::new(inner, 0)); let retrying = RetryingObjectStore::new(flaky.clone(), test_rand(), test_clock(), None); let path = Path::from("/data/obj"); @@ -743,7 +743,7 @@ mod tests { #[tokio::test] async fn test_head_retries_transient_until_success() { - let inner: Arc = Arc::new(InMemory::new()); + let inner: Arc = Arc::new(InMemory::new()); let path = Path::from("/x"); inner .put(&path, PutPayload::from_bytes(Bytes::from_static(b"data"))) @@ -760,7 +760,7 @@ mod tests { #[tokio::test] async fn test_put_opts_does_not_retry_on_precondition() { - let inner: Arc = Arc::new(InMemory::new()); + let inner: Arc = Arc::new(InMemory::new()); let failing = Arc::new(FlakyObjectStore::new(inner, 0).with_put_precondition_always()); let retrying = RetryingObjectStore::new(failing.clone(), test_rand(), test_clock(), None); let path = Path::from("/p"); @@ -783,7 +783,7 @@ mod tests { #[tokio::test] async fn test_get_opts_does_not_retry_on_not_modified() { - let inner: Arc = Arc::new(InMemory::new()); + let inner: Arc = Arc::new(InMemory::new()); let retrying = RetryingObjectStore::new(inner.clone(), test_rand(), test_clock(), None); let path = Path::from("/data/obj"); @@ -812,7 +812,7 @@ mod tests { #[tokio::test] async fn test_list_retries_transient_until_success() { - let inner: Arc = Arc::new(InMemory::new()); + let inner: Arc = Arc::new(InMemory::new()); let paths = [ Path::from("/items/a"), Path::from("/items/b"), @@ -847,7 +847,7 @@ mod tests { #[tokio::test] async fn test_list_with_offset_retries_transient_until_success() { - let inner: Arc = Arc::new(InMemory::new()); + let inner: Arc = Arc::new(InMemory::new()); let paths = [ Path::from("/items/a"), Path::from("/items/b"), @@ -885,7 +885,7 @@ mod tests { async fn test_put_opts_succeeds_on_matching_ulid() { // Simulate: put succeeds but returns AlreadyExists error (timeout after write) // The ULID in the object's metadata should match, so we return success - let inner: Arc = Arc::new(InMemory::new()); + let inner: Arc = Arc::new(InMemory::new()); let flaky = Arc::new( FlakyObjectStore::new(inner, 0).with_put_succeeds_but_returns_already_exists(), ); @@ -910,7 +910,7 @@ mod tests { #[tokio::test] async fn test_put_opts_fails_on_mismatched_ulid() { // First write a file with different ULID (simulating another client's write) - let inner: Arc = Arc::new(InMemory::new()); + let inner: Arc = Arc::new(InMemory::new()); let path = Path::from("/data/obj"); // Write directly to inner store (no ULID from RetryingObjectStore) @@ -948,7 +948,7 @@ mod tests { #[tokio::test] async fn test_get_range_retries_transient_until_success() { - let inner: Arc = Arc::new(InMemory::new()); + let inner: Arc = Arc::new(InMemory::new()); let path = Path::from("/data/obj"); inner .put( @@ -972,7 +972,7 @@ mod tests { #[tokio::test] async fn test_get_ranges_retries_transient_until_success() { - let inner: Arc = Arc::new(InMemory::new()); + let inner: Arc = Arc::new(InMemory::new()); let path = Path::from("/data/obj"); inner .put( @@ -1001,7 +1001,7 @@ mod tests { use object_store::{Attribute, Attributes, GetOptions}; use std::borrow::Cow; - let inner: Arc = Arc::new(InMemory::new()); + let inner: Arc = Arc::new(InMemory::new()); let retrying = RetryingObjectStore::new(inner.clone(), test_rand(), test_clock(), None); let path = Path::from("/data/obj"); @@ -1120,7 +1120,7 @@ mod tests { #[tokio::test] async fn test_bounded_max_retries_gives_up_instead_of_retrying_forever() { // Store fails more times (5) than the configured retry bound (2). - let inner: Arc = Arc::new(InMemory::new()); + let inner: Arc = Arc::new(InMemory::new()); let flaky = Arc::new(FlakyObjectStore::new(inner, 5)); let retrying = RetryingObjectStore::new(flaky.clone(), test_rand(), test_clock(), Some(2)); diff --git a/slatedb/src/size_tiered_compaction.rs b/slatedb/src/size_tiered_compaction.rs index 52fd63d3b2..0dd3c22a08 100644 --- a/slatedb/src/size_tiered_compaction.rs +++ b/slatedb/src/size_tiered_compaction.rs @@ -264,7 +264,7 @@ impl CompactionScheduler for SizeTieredCompactionScheduler { &self, state: &CompactorStateView, compaction: &CompactionSpec, - ) -> Result<(), crate::error::Error> { + ) -> Result<(), Error> { // Size-tiered does not propose drain specs and has no policy // opinions on them. Drain invariants belong to the compactor-level // validation. diff --git a/slatedb/src/sst_builder.rs b/slatedb/src/sst_builder.rs index edc5853e8b..65b25119d1 100644 --- a/slatedb/src/sst_builder.rs +++ b/slatedb/src/sst_builder.rs @@ -1389,7 +1389,7 @@ mod tests { let transformer = Arc::new(XorTransformer { key: 0xAB }); #[cfg(feature = "snappy")] - let compression = Some(crate::config::CompressionCodec::Snappy); + let compression = Some(CompressionCodec::Snappy); #[cfg(not(feature = "snappy"))] let compression = None; diff --git a/slatedb/src/sst_iter.rs b/slatedb/src/sst_iter.rs index ff3855d201..edee72f77e 100644 --- a/slatedb/src/sst_iter.rs +++ b/slatedb/src/sst_iter.rs @@ -82,7 +82,7 @@ impl SstView<'_> { fn point_key(&self) -> Option<&[u8]> { match (self.start_key(), self.end_key()) { - (Bound::Included(start), Bound::Included(end)) if start == end => Some(start), + (Included(start), Included(end)) if start == end => Some(start), _ => None, } } diff --git a/slatedb/src/sst_reader.rs b/slatedb/src/sst_reader.rs index c880d5e719..c5441512f0 100644 --- a/slatedb/src/sst_reader.rs +++ b/slatedb/src/sst_reader.rs @@ -587,11 +587,7 @@ mod tests { let store: Arc = Arc::new(InMemory::new()); let reader = SstReader::new("/test", store, None, None); - let wal_handle = SsTableHandle::new( - SsTableId::Wal(42), - 0, - crate::db_state::SsTableInfo::default(), - ); + let wal_handle = SsTableHandle::new(SsTableId::Wal(42), 0, SsTableInfo::default()); let result = reader.open_with_handle(wal_handle); assert!(result.is_err()); } diff --git a/slatedb/src/sst_stats.rs b/slatedb/src/sst_stats.rs index 7fe3d9dce9..727bda2712 100644 --- a/slatedb/src/sst_stats.rs +++ b/slatedb/src/sst_stats.rs @@ -40,7 +40,7 @@ impl SstStats { /// Returns the in-memory size in bytes (struct + heap-allocated block_stats). pub(crate) fn size(&self) -> usize { - std::mem::size_of::() + self.block_stats.len() * std::mem::size_of::() + size_of::() + self.block_stats.len() * size_of::() } /// Returns a clone. diff --git a/slatedb/src/tablestore.rs b/slatedb/src/tablestore.rs index 7a0717264d..5a5b16e32b 100644 --- a/slatedb/src/tablestore.rs +++ b/slatedb/src/tablestore.rs @@ -361,13 +361,13 @@ impl TableStore { self.fp_registry.clone(), "write-wal-sst-io-error", matches!(id, SsTableId::Wal(_)), - |_| Result::Err(slatedb_io_error()) + |_| Err(slatedb_io_error()) ); fail_point!( self.fp_registry.clone(), "write-compacted-sst-io-error", matches!(id, SsTableId::Compacted(_)), - |_| Result::Err(slatedb_io_error()) + |_| Err(slatedb_io_error()) ); let object_store = self.object_stores.store_for(id); @@ -483,7 +483,7 @@ impl TableStore { pub(crate) async fn write_wal_fence(&self, wal_id: u64) -> Result<(), SlateDBError> { let id = SsTableId::Wal(wal_id); fail_point!(self.fp_registry.clone(), "write-wal-sst-io-error", |_| { - Result::Err(slatedb_io_error()) + Err(slatedb_io_error()) }); write_sst_in_object_store( self.object_stores.store_for(&id), @@ -2549,7 +2549,7 @@ mod tests { // Create id1, id2, and i3 as three random UUIDs that have been sorted ascending. // Need to do this because the Ulids are sometimes generated in the same millisecond // and the random suffix is used to break the tie, which might be out of order. - let mut ulids = (0..3).map(|_| ulid::Ulid::new()).collect::>(); + let mut ulids = (0..3).map(|_| Ulid::new()).collect::>(); ulids.sort(); let (id1, id2, id3) = ( SsTableId::Compacted(ulids[0]), diff --git a/slatedb/src/test_utils.rs b/slatedb/src/test_utils.rs index 4d3ebb6daa..c6d3573b7c 100644 --- a/slatedb/src/test_utils.rs +++ b/slatedb/src/test_utils.rs @@ -725,7 +725,7 @@ impl ObjectStore for FlakyObjectStore { &self, location: &Path, options: GetOptions, - ) -> object_store::Result { + ) -> object_store::Result { if options.head { self.head_attempts.fetch_add(1, Ordering::SeqCst); if self @@ -788,9 +788,9 @@ impl ObjectStore for FlakyObjectStore { let extensions = result.extensions.clone(); let body = result.bytes().await?; let truncated = body.slice(..truncate_bytes.min(body.len())); - return Ok(object_store::GetResult { + return Ok(GetResult { payload: object_store::GetResultPayload::Stream( - futures::stream::once(async { Ok(truncated) }).boxed(), + stream::once(async { Ok(truncated) }).boxed(), ), meta, range, @@ -868,7 +868,7 @@ impl ObjectStore for FlakyObjectStore { async fn put_multipart_opts( &self, location: &Path, - opts: object_store::PutMultipartOptions, + opts: PutMultipartOptions, ) -> object_store::Result> { self.put_multipart_attempts.fetch_add(1, Ordering::SeqCst); self.inner.put_multipart_opts(location, opts).await @@ -1148,7 +1148,7 @@ impl ObjectStore for GatedObjectStore { &self, location: &Path, options: GetOptions, - ) -> object_store::Result { + ) -> object_store::Result { if options.head { self.head_gate.wait().await?; } else { @@ -1170,7 +1170,7 @@ impl ObjectStore for GatedObjectStore { async fn put_multipart_opts( &self, location: &Path, - opts: object_store::PutMultipartOptions, + opts: PutMultipartOptions, ) -> object_store::Result> { self.put_multipart_opts_gate.wait().await?; self.inner.put_multipart_opts(location, opts).await @@ -1224,7 +1224,7 @@ impl ObjectStore for GatedObjectStore { &self, from: &Path, to: &Path, - options: object_store::RenameOptions, + options: RenameOptions, ) -> object_store::Result<()> { self.rename_gate.wait().await?; self.inner.rename_opts(from, to, options).await @@ -1659,7 +1659,7 @@ impl ObjectStore for RecordingObjectStore { &self, location: &Path, options: GetOptions, - ) -> object_store::Result { + ) -> object_store::Result { let tag = ObjectStoreCallTag::from_extensions(&options.extensions); self.calls.lock().push(RecordedCall::Get { head: options.head, @@ -1687,7 +1687,7 @@ impl ObjectStore for RecordingObjectStore { async fn put_multipart_opts( &self, location: &Path, - opts: object_store::PutMultipartOptions, + opts: PutMultipartOptions, ) -> object_store::Result> { let tag = ObjectStoreCallTag::from_extensions(&opts.extensions); self.calls.lock().push(RecordedCall::PutMultipart { diff --git a/slatedb/src/types.rs b/slatedb/src/types.rs index d4306b7b76..8eddcdd328 100644 --- a/slatedb/src/types.rs +++ b/slatedb/src/types.rs @@ -48,13 +48,13 @@ impl RowEntry { pub(crate) fn estimated_size(&self) -> usize { let mut size = self.key.len() + self.value.len(); // Add size for sequence number - size += std::mem::size_of::(); + size += size_of::(); // Add size for timestamps if self.create_ts.is_some() { - size += std::mem::size_of::(); + size += size_of::(); } if self.expire_ts.is_some() { - size += std::mem::size_of::(); + size += size_of::(); } size } @@ -63,20 +63,20 @@ impl RowEntry { /// The `key_prefix_len` is the number of bytes shared with the block's first key. pub(crate) fn encoded_size(&self, key_prefix_len: usize) -> usize { let key_suffix_len = self.key.len() - key_prefix_len; - let mut size = std::mem::size_of::() // key_prefix_len - + std::mem::size_of::() // key_suffix_len + let mut size = size_of::() // key_prefix_len + + size_of::() // key_suffix_len + key_suffix_len - + std::mem::size_of::() // seq - + std::mem::size_of::(); // flags + + size_of::() // seq + + size_of::(); // flags if self.expire_ts.is_some() { - size += std::mem::size_of::(); + size += size_of::(); } if self.create_ts.is_some() { - size += std::mem::size_of::(); + size += size_of::(); } if !self.value.is_tombstone() { - size += std::mem::size_of::(); // value_len + size += size_of::(); // value_len size += self.value.len(); } size diff --git a/slatedb/src/utils.rs b/slatedb/src/utils.rs index e7d16ef1b2..2777b2063b 100644 --- a/slatedb/src/utils.rs +++ b/slatedb/src/utils.rs @@ -454,7 +454,7 @@ where I::Item: Send, T: Send, F: Fn(I::Item) -> Fut + Send, - Fut: std::future::Future, SlateDBError>> + Send, + Fut: Future, SlateDBError>> + Send, { let mut out = VecDeque::new(); @@ -522,11 +522,8 @@ pub(crate) fn panic_string(panic: &Box) -> String { /// - (Err(SlateDBError::BackgroundTaskPanic), Some(payload)) if the task panicked pub(crate) fn split_unwind_result( name: String, - unwind_result: Result, Box>, -) -> ( - Result<(), SlateDBError>, - Option>, -) { + unwind_result: Result, Box>, +) -> (Result<(), SlateDBError>, Option>) { match unwind_result { Ok(result) => (result, None), Err(payload) => (Err(SlateDBError::BackgroundTaskPanic(name)), Some(payload)), @@ -552,10 +549,7 @@ pub(crate) fn split_unwind_result( pub(crate) fn split_join_result( name: String, join_result: Result, tokio::task::JoinError>, -) -> ( - Result<(), SlateDBError>, - Option>, -) { +) -> (Result<(), SlateDBError>, Option>) { match join_result { Ok(task_result) => (task_result, None), Err(join_error) => { @@ -1485,8 +1479,7 @@ mod tests { #[test] fn test_split_unwind_result_ok_ok() { // Given: a successful unwind result - let unwind_result: Result, Box> = - Ok(Ok(())); + let unwind_result: Result, Box> = Ok(Ok(())); // When: we split the result let (result, payload) = super::split_unwind_result("test".to_string(), unwind_result); @@ -1499,7 +1492,7 @@ mod tests { #[test] fn test_split_unwind_result_ok_error() { // Given: an unwind result with a task error - let unwind_result: Result, Box> = + let unwind_result: Result, Box> = Ok(Err(SlateDBError::Fenced)); // When: we split the result @@ -1514,7 +1507,7 @@ mod tests { fn test_split_unwind_result_panic() { // Given: an unwind result that panicked with a non-SlateDBError (e.g., a string) let panic_msg = "something went wrong"; - let unwind_result: Result, Box> = + let unwind_result: Result, Box> = Err(Box::new(panic_msg)); // When: we split the result diff --git a/slatedb/src/wal/wal_sst_builder.rs b/slatedb/src/wal/wal_sst_builder.rs index 758a56984a..f7cac6dd71 100644 --- a/slatedb/src/wal/wal_sst_builder.rs +++ b/slatedb/src/wal/wal_sst_builder.rs @@ -868,7 +868,7 @@ mod tests { let encoded = builder.build().await.unwrap(); // then: - assert_eq!(encoded.info.sst_type, crate::db_state::SstType::Wal,); + assert_eq!(encoded.info.sst_type, SstType::Wal,); } mod block_transformer_tests { diff --git a/slatedb/src/wal_buffer.rs b/slatedb/src/wal_buffer.rs index c0051409c8..b0fe873b39 100644 --- a/slatedb/src/wal_buffer.rs +++ b/slatedb/src/wal_buffer.rs @@ -502,7 +502,7 @@ impl WalFlushHandler { warn!("outstanding references to wal id {} after flushing", wal_id); } drop(wal); - self.notify_listener(wal::WalEvent::WalFlushed(status)); + self.notify_listener(WalEvent::WalFlushed(status)); } Ok(()) @@ -526,7 +526,7 @@ impl WalFlushHandler { Ok(()) } - fn notify_listener(&self, event: wal::WalEvent) { + fn notify_listener(&self, event: WalEvent) { if let Some(l) = self.listener.as_ref() { (*l)(event); } @@ -863,7 +863,7 @@ mod tests { observer .subscribe(Arc::new(move |status| { (*listener)(status.clone()); - let wal::WalEvent::WalFlushed(status) = status else { + let WalEvent::WalFlushed(status) = status else { return; }; oracle.advance_durable_seq(status.last_flushed_seq.unwrap_or(0)) @@ -1011,8 +1011,8 @@ mod tests { ); } - fn recording_listener() -> (wal::WalStatusListener, Arc>>) { - let events = Arc::new(std::sync::Mutex::new(Vec::new())); + fn recording_listener() -> (wal::WalStatusListener, Arc>>) { + let events = Arc::new(Mutex::new(Vec::new())); let recorder = events.clone(); let listener = Arc::new(move |event| { recorder.lock().unwrap().push(event); @@ -1038,7 +1038,7 @@ mod tests { let mut flushed: Vec<_> = recorded .iter() .filter_map(|e| { - let wal::WalEvent::WalFlushed(status) = e else { + let WalEvent::WalFlushed(status) = e else { return None; }; Some(status) From 8cc31a54fd7c94149b4290c47a5431dff2dd9973 Mon Sep 17 00:00:00 2001 From: Tyler Rockwood Date: Sat, 15 Aug 2026 17:49:03 -0500 Subject: [PATCH 33/65] example: Add a rescaling example (#2028) --- examples/Cargo.toml | 6 + examples/src/rescaling.rs | 402 ++++++++++++++++++++++++++++++++++++++ 2 files changed, 408 insertions(+) create mode 100644 examples/src/rescaling.rs diff --git a/examples/Cargo.toml b/examples/Cargo.toml index 19dc78ad3d..755dd6f2e4 100644 --- a/examples/Cargo.toml +++ b/examples/Cargo.toml @@ -45,6 +45,12 @@ path = "src/refresh_checkpoint.rs" test = false bench = false +[[bin]] +name = "rescaling" +path = "src/rescaling.rs" +test = false +bench = false + [[bin]] name = "range-scans" path = "src/range_scans.rs" diff --git a/examples/src/rescaling.rs b/examples/src/rescaling.rs new file mode 100644 index 0000000000..10a8fc2c73 --- /dev/null +++ b/examples/src/rescaling.rs @@ -0,0 +1,402 @@ +//! Rescaling a database by splitting and merging key ranges. +//! +//! Scale-up (split) projects one source into two clones with disjoint key +//! ranges. Scale-down (merge) unions those clones back into one database. +//! Both are O(1) manifest views over shared SSTs — no SST data is copied. +//! +//! Projection and union reject sources that still have data in the WAL, so +//! this example flushes the WAL and memtables into L0, then checkpoints with +//! [`CheckpointScope::Durable`] before cloning. The same APIs work for +//! segmented and non-segmented stores. +//! +//! Clone construction uses: +//! [`AdminBuilder::new`] → [`Admin::create_clone_builder_from_source`] → +//! [`CloneBuilder::with_source`] → [`CloneBuilder::build`]. + +use slatedb::admin::{AdminBuilder, CloneSourceSpec}; +use slatedb::bytes::Bytes; +use slatedb::config::{CheckpointOptions, CheckpointScope, FlushOptions, FlushType}; +use slatedb::object_store::memory::InMemory; +use slatedb::{CheckpointCreateResult, Db, Error, PrefixExtractor, PrefixTarget}; +use std::ops::{Bound, RangeBounds}; +use std::sync::Arc; + +type ProjectionRange = (Bound, Bound); + +/// Tenant IDs sort as bytes, so zoos `< "metro"` land on the left shard. +const SPLIT_TENANT: &[u8] = b"metro"; + +fn left_tenants() -> ProjectionRange { + ( + Bound::Unbounded, + Bound::Excluded(Bytes::from_static(SPLIT_TENANT)), + ) +} + +fn right_tenants() -> ProjectionRange { + ( + Bound::Included(Bytes::from_static(SPLIT_TENANT)), + Bound::Unbounded, + ) +} + +/// Two LSM segments — bulky animal records vs a smaller owner index. +/// +/// Keys are kind-first (`data/…`, `idx/…`) so the extractor names the segments +/// `data` and `idx`. Tenants live in the next path component, so each zoo's +/// rows are not one contiguous byte range; tenant splits use +/// [`CloneBuilder::with_segment_projection`]. +struct DataIdxSegmentExtractor; + +impl PrefixExtractor for DataIdxSegmentExtractor { + fn name(&self) -> &str { + "data_idx" + } + + fn prefix_len(&self, target: &PrefixTarget) -> Option { + let key = match target { + PrefixTarget::Point(key) | PrefixTarget::Prefix(key) => key.as_ref(), + }; + if key == b"data" || key.starts_with(b"data/") { + Some(b"data".len()) + } else if key == b"idx" || key.starts_with(b"idx/") { + Some(b"idx".len()) + } else { + None + } + } +} + +fn animal_key(zoo: &[u8], animal_id: &[u8]) -> Vec { + [b"data/", zoo, b"/animal/", animal_id].concat() +} + +fn owner_index_key(zoo: &[u8], owner: &[u8], animal_id: &[u8]) -> Vec { + [b"idx/", zoo, b"/owner/", owner, b"/", animal_id].concat() +} + +fn kind_tenant(kind: &[u8], tenant: &[u8]) -> Bytes { + Bytes::from([kind, b"/", tenant].concat()) +} + +/// Per-segment view of zoos `< metro` (valid inside `[prefix, prefix++)`). +fn left_tenant_in_segment(prefix: &[u8]) -> ProjectionRange { + ( + Bound::Unbounded, + Bound::Excluded(kind_tenant(prefix, SPLIT_TENANT)), + ) +} + +/// Per-segment view of zoos `>= metro`. +fn right_tenant_in_segment(prefix: &[u8]) -> ProjectionRange { + ( + Bound::Included(kind_tenant(prefix, SPLIT_TENANT)), + Bound::Unbounded, + ) +} + +/// Contiguous `data/…` / `idx/…` slices for union. A tenant shard that holds +/// both segments has a bounding range spanning `data/…` through `idx/…`, so +/// left and right overlap unless re-sliced before merge. +fn data_left_range() -> ProjectionRange { + ( + Bound::Unbounded, + Bound::Excluded(Bytes::from_static(b"idx")), + ) +} + +fn data_right_range() -> ProjectionRange { + ( + Bound::Included(kind_tenant(b"data", SPLIT_TENANT)), + Bound::Excluded(Bytes::from_static(b"idx")), + ) +} + +fn idx_left_range() -> ProjectionRange { + ( + Bound::Included(Bytes::from_static(b"idx")), + Bound::Excluded(kind_tenant(b"idx", SPLIT_TENANT)), + ) +} + +fn idx_right_range() -> ProjectionRange { + ( + Bound::Included(kind_tenant(b"idx", SPLIT_TENANT)), + Bound::Unbounded, + ) +} + +#[tokio::main] +async fn main() -> anyhow::Result<()> { + let object_store = Arc::new(InMemory::new()); + + println!("=== Non-segmented rescaling ==="); + rescale_non_segmented(object_store.clone()).await?; + + println!("\n=== Segmented rescaling (data + idx segments, split by zoo) ==="); + rescale_segmented(object_store).await?; + + Ok(()) +} + +async fn rescale_non_segmented(object_store: Arc) -> anyhow::Result<()> { + let root_path = "/tmp/slatedb_rescaling/plain/root"; + let left_path = "/tmp/slatedb_rescaling/plain/left"; + let right_path = "/tmp/slatedb_rescaling/plain/right"; + let merged_path = "/tmp/slatedb_rescaling/plain/merged"; + + // Tenant-prefixed keys without a segment extractor — still split by zoo. + let db = Db::open(root_path, object_store.clone()).await?; + db.put(b"bronx/lion", b"Leo").await?; + db.put(b"lincoln/otter", b"Ollie").await?; + db.put(b"metro/panda", b"Mei").await?; + db.put(b"oakland/zebra", b"Ziggy").await?; + let checkpoint = checkpoint_for_rescale(&db).await?; + db.close().await?; + + create_clone( + left_path, + vec![CloneSourceSpec::with_checkpoint(root_path, checkpoint.id) + .with_projection_range(left_tenants())], + object_store.clone(), + ) + .await?; + create_clone( + right_path, + vec![CloneSourceSpec::with_checkpoint(root_path, checkpoint.id) + .with_projection_range(right_tenants())], + object_store.clone(), + ) + .await?; + + let left = Db::open(left_path, object_store.clone()).await?; + let right = Db::open(right_path, object_store.clone()).await?; + assert_eq!( + left.get(b"bronx/lion").await?, + Some(b"Leo".as_slice().into()) + ); + assert_eq!(left.get(b"metro/panda").await?, None); + assert_eq!( + right.get(b"metro/panda").await?, + Some(b"Mei".as_slice().into()) + ); + assert_eq!(right.get(b"bronx/lion").await?, None); + println!("split by tenant: left has bronx/lincoln; right has metro/oakland"); + left.close().await?; + right.close().await?; + + create_clone( + merged_path, + vec![ + CloneSourceSpec::new(left_path).with_projection_range(left_tenants()), + CloneSourceSpec::new(right_path).with_projection_range(right_tenants()), + ], + object_store.clone(), + ) + .await?; + + let merged = Db::open(merged_path, object_store).await?; + assert_eq!( + merged.get(b"bronx/lion").await?, + Some(b"Leo".as_slice().into()) + ); + assert_eq!( + merged.get(b"oakland/zebra").await?, + Some(b"Ziggy".as_slice().into()) + ); + println!("merged: all zoos are visible again"); + merged.close().await?; + + Ok(()) +} + +async fn rescale_segmented(object_store: Arc) -> anyhow::Result<()> { + let extractor = Arc::new(DataIdxSegmentExtractor); + let root_path = "/tmp/slatedb_rescaling/segmented/root"; + let left_path = "/tmp/slatedb_rescaling/segmented/left"; + let right_path = "/tmp/slatedb_rescaling/segmented/right"; + let merged_data_path = "/tmp/slatedb_rescaling/segmented/merged_data"; + let merged_idx_path = "/tmp/slatedb_rescaling/segmented/merged_idx"; + let merged_path = "/tmp/slatedb_rescaling/segmented/merged"; + + let db = Db::builder(root_path, object_store.clone()) + .with_segment_extractor(extractor.clone()) + .build() + .await?; + + // bronx + lincoln → left of the split; metro + oakland → right. + put_animal(&db, b"bronx", b"lion-1", b"alice", b"Leo the lion").await?; + put_animal(&db, b"lincoln", b"otter-1", b"bob", b"Ollie the otter").await?; + put_animal(&db, b"metro", b"panda-1", b"carol", b"Mei the panda").await?; + put_animal(&db, b"oakland", b"zebra-1", b"dave", b"Ziggy the zebra").await?; + + let checkpoint = checkpoint_for_rescale(&db).await?; + db.close().await?; + + // Scale up: keep each zoo's data + owner-index together via per-segment + // projection (`data/{zoo}/…` and `idx/{zoo}/…` are not one byte range). + create_clone_with_segment_projection( + left_path, + CloneSourceSpec::with_checkpoint(root_path, checkpoint.id), + object_store.clone(), + left_tenant_in_segment, + ) + .await?; + create_clone_with_segment_projection( + right_path, + CloneSourceSpec::with_checkpoint(root_path, checkpoint.id), + object_store.clone(), + right_tenant_in_segment, + ) + .await?; + + let left = Db::builder(left_path, object_store.clone()) + .with_segment_extractor(extractor.clone()) + .build() + .await?; + let right = Db::builder(right_path, object_store.clone()) + .with_segment_extractor(extractor.clone()) + .build() + .await?; + + assert_eq!( + left.get(animal_key(b"bronx", b"lion-1")).await?, + Some(b"Leo the lion".as_slice().into()) + ); + assert_eq!( + left.get(owner_index_key(b"bronx", b"alice", b"lion-1")) + .await?, + Some(Bytes::new()) + ); + assert_eq!(left.get(animal_key(b"metro", b"panda-1")).await?, None); + assert_eq!( + left.get(owner_index_key(b"metro", b"carol", b"panda-1")) + .await?, + None + ); + assert_eq!( + right.get(animal_key(b"metro", b"panda-1")).await?, + Some(b"Mei the panda".as_slice().into()) + ); + assert_eq!( + right + .get(owner_index_key(b"metro", b"carol", b"panda-1")) + .await?, + Some(Bytes::new()) + ); + assert_eq!(right.get(animal_key(b"bronx", b"lion-1")).await?, None); + println!("split by tenant: left has bronx/lincoln (data+idx); right has metro/oakland"); + left.close().await?; + right.close().await?; + + // Scale down: union needs non-overlapping bounding ranges, so re-slice each + // shard into contiguous `data/…` and `idx/…` halves, then union those. + create_clone( + merged_data_path, + vec![ + CloneSourceSpec::new(left_path).with_projection_range(data_left_range()), + CloneSourceSpec::new(right_path).with_projection_range(data_right_range()), + ], + object_store.clone(), + ) + .await?; + create_clone( + merged_idx_path, + vec![ + CloneSourceSpec::new(left_path).with_projection_range(idx_left_range()), + CloneSourceSpec::new(right_path).with_projection_range(idx_right_range()), + ], + object_store.clone(), + ) + .await?; + create_clone( + merged_path, + vec![ + CloneSourceSpec::new(merged_data_path), + CloneSourceSpec::new(merged_idx_path), + ], + object_store.clone(), + ) + .await?; + + let merged = Db::builder(merged_path, object_store) + .with_segment_extractor(extractor) + .build() + .await?; + assert_eq!( + merged.get(animal_key(b"bronx", b"lion-1")).await?, + Some(b"Leo the lion".as_slice().into()) + ); + assert_eq!( + merged + .get(owner_index_key(b"oakland", b"dave", b"zebra-1")) + .await?, + Some(Bytes::new()) + ); + println!("merged: all zoo data and owner-index rows are visible again"); + merged.close().await?; + + Ok(()) +} + +async fn put_animal( + db: &Db, + zoo: &[u8], + animal_id: &[u8], + owner: &[u8], + record: &[u8], +) -> Result<(), Error> { + db.put(animal_key(zoo, animal_id), record).await?; + db.put(owner_index_key(zoo, owner, animal_id), b"").await?; + Ok(()) +} + +/// Build a clone from one or more sources. +async fn create_clone( + clone_path: &str, + sources: Vec>, + object_store: Arc, +) -> Result<(), Error> { + let admin = AdminBuilder::new(clone_path, object_store).build(); + let mut sources = sources.into_iter(); + let first = sources + .next() + .expect("rescaling clone requires at least one source"); + let mut builder = admin.create_clone_builder_from_source(first); + for source in sources { + builder = builder.with_source(source); + } + builder.build().await +} + +/// Build a single-source clone with a per-segment projection. +async fn create_clone_with_segment_projection( + clone_path: &str, + source: CloneSourceSpec, + object_store: Arc, + segment_projection: F, +) -> Result<(), Error> +where + F: Fn(&[u8]) -> R + Send + Sync + 'static, + R: RangeBounds, +{ + AdminBuilder::new(clone_path, object_store) + .build() + .create_clone_builder_from_source(source) + .with_segment_projection(segment_projection) + .build() + .await +} + +/// Flush every write into L0, then pin that state. Projection and union reject +/// sources that still have non-empty WAL SSTs. +async fn checkpoint_for_rescale(db: &Db) -> anyhow::Result { + db.flush().await?; + db.flush_with_options(FlushOptions { + flush_type: FlushType::MemTable, + }) + .await?; + Ok(db + .create_checkpoint(CheckpointScope::Durable, &CheckpointOptions::default()) + .await?) +} From e98a897f117a31aef25e5030534255445f426360 Mon Sep 17 00:00:00 2001 From: Chris Date: Mon, 17 Aug 2026 09:19:02 -0700 Subject: [PATCH 34/65] Add benchmark.slatedb.io blog post (#2026) --- .../benchmark-balanced-get-latency.html | 166 ++++++++++++ .../benchmark-sustained-ingest-put.html | 138 ++++++++++ .../components/BenchmarkComparisonTable.astro | 237 ++++++++++++++++++ .../src/content/blog/benchmarking-slatedb.mdx | 86 +++++++ 4 files changed, 627 insertions(+) create mode 100644 website/public/charts/benchmark-balanced-get-latency.html create mode 100644 website/public/charts/benchmark-sustained-ingest-put.html create mode 100644 website/src/components/BenchmarkComparisonTable.astro create mode 100644 website/src/content/blog/benchmarking-slatedb.mdx diff --git a/website/public/charts/benchmark-balanced-get-latency.html b/website/public/charts/benchmark-balanced-get-latency.html new file mode 100644 index 0000000000..95daba56e2 --- /dev/null +++ b/website/public/charts/benchmark-balanced-get-latency.html @@ -0,0 +1,166 @@ + + + + + +SlateDB balanced workload get application latency + + + +

+

Application latency · get

+
+ avg + p50 + p99 + p99.9 + full benchmark result ↗ +
+
+ + SlateDB balanced workload get application latency + Average and percentile get latency over the balanced workload measurement, with prominent spikes in the p99 and p99.9 series. + +
+
+
+ + + diff --git a/website/public/charts/benchmark-sustained-ingest-put.html b/website/public/charts/benchmark-sustained-ingest-put.html new file mode 100644 index 0000000000..c6c5754500 --- /dev/null +++ b/website/public/charts/benchmark-sustained-ingest-put.html @@ -0,0 +1,138 @@ + + + + + +SlateDB sustained-ingest application throughput + + + +
+

Application throughput

+
+ put + published average (91.58 MiB/s) + full benchmark result ↗ +
+
+ + SlateDB sustained-ingest application put throughput + One-second application put throughput over roughly twenty minutes, with repeated drops as SlateDB throttles writes. The published average is 91.58 MiB per second. + +
+
+
+ + + diff --git a/website/src/components/BenchmarkComparisonTable.astro b/website/src/components/BenchmarkComparisonTable.astro new file mode 100644 index 0000000000..7ed5cf15a4 --- /dev/null +++ b/website/src/components/BenchmarkComparisonTable.astro @@ -0,0 +1,237 @@ +--- +type Support = boolean | 'manual'; + +const benchmarks = ['YCSB', 'KVBench', 'db_bench', 'Tectonic']; +const groups: Array<{ + label: string; + rows: Array<{ capability: string; support: Support[] }>; +}> = [ + { + label: 'Operations', + rows: [ + { capability: 'Insert', support: [true, true, true, true] }, + { capability: 'Update', support: [true, true, true, true] }, + { capability: 'Read-modify-write', support: [true, false, true, true] }, + { capability: 'Point query', support: [true, true, true, true] }, + { capability: 'Empty point query', support: [false, true, true, true] }, + { capability: 'Range query', support: [true, true, true, true] }, + { capability: 'Point delete', support: [false, true, true, true] }, + { capability: 'Empty point delete', support: [false, true, false, true] }, + { capability: 'Range delete', support: [false, true, true, true] }, + ], + }, + { + label: 'Distributions', + rows: [ + { capability: 'Uniform', support: [true, true, true, true] }, + { capability: 'Normal', support: [false, true, true, true] }, + { capability: 'Beta', support: [false, true, false, true] }, + { capability: 'Zipfian', support: [true, true, false, true] }, + { capability: 'Exponential', support: [false, false, true, true] }, + { capability: 'Log normal', support: [false, false, false, true] }, + { capability: 'Poisson', support: [false, false, false, true] }, + { capability: 'Weibull', support: [false, false, false, true] }, + { capability: 'Pareto', support: [false, false, true, true] }, + ], + }, + { + label: 'Properties', + rows: [ + { capability: 'Dynamic workload shifts', support: [false, 'manual', 'manual', true] }, + { capability: 'Context-aware shifting', support: [false, false, false, true] }, + { capability: 'Data sortedness', support: [false, false, false, true] }, + { capability: 'Variable query selectivity', support: [true, false, true, true] }, + { capability: 'Variable key-value length', support: [false, false, false, true] }, + { capability: 'Temporality-based access', support: [false, false, true, true] }, + { capability: 'Customizable key prefix', support: [false, false, 'manual', true] }, + { capability: 'Composite keys', support: [false, false, false, true] }, + ], + }, +]; +--- + +
+
+ + + + + + + + + + + + {benchmarks.map((benchmark) => ( + + ))} + + + {groups.map((group) => ( + + {group.rows.map((row, rowIndex) => ( + + {rowIndex === 0 && ( + + )} + + {row.support.map((value) => ( + + ))} + + ))} + + ))} +
Feature coverage across YCSB, KVBench, db_bench, and Tectonic
CategoryCapability{benchmark}
+ {group.label} + {row.capability} + {value === true ? ( + <>Supported + ) : value === 'manual' ? ( + <>Requires significant manual intervention + ) : ( + Not supported + )} +
+
+
* Requires significant manual intervention.
+
+ + diff --git a/website/src/content/blog/benchmarking-slatedb.mdx b/website/src/content/blog/benchmarking-slatedb.mdx new file mode 100644 index 0000000000..0655c21143 --- /dev/null +++ b/website/src/content/blog/benchmarking-slatedb.mdx @@ -0,0 +1,86 @@ +--- +title: "Benchmarking SlateDB" +pubDate: 2026-08-12 +author: Chris +authorGithub: criccomini +# ogImage: /img/some-custom-card.jpg # optional per-post override +--- + +import ChartEmbed from '../../components/ChartEmbed.astro'; +import BenchmarkComparisonTable from '../../components/BenchmarkComparisonTable.astro'; + +SlateDB has been around for a few years now. We've begun to pay more attention to performance lately. As [Kent Beck](https://en.wikipedia.org/wiki/Kent_Beck) (purportedly) said, ["Make It Work, Make It Right, Make It Fast."](https://wiki.c2.com/?MakeItWorkMakeItRightMakeItFast) We feel we've earned the right to make SlateDB fast. Of course, before we improve performance, we must measure it. That means benchmarks. + +Benchmarks are tricky. It's easy to do something that is perceived as marketing, hyperbole, or outright dishonesty. It's questionable whether public benchmarks are even useful anymore. A frontier LLM is capable of building excellent bespoke benchmarks for each user. + +Yet we found ourselves rebuilding SlateDB's benchmark suite a few weeks back. We--the SlateDB developers--still need them to measure performance. We also believe the data will help users gauge whether SlateDB meets their performance needs. + +This post covers the benchmarking philosophy we landed on, how we benchmarked SlateDB, and some things we learned along the way. + +(You can head over to [benchmark.slatedb.io](https://benchmark.slatedb.io) if you just want the results.) + +## Legacy benchmarks + +We have had benchmarks for quite a while: both [Criterion microbenchmarks](https://github.com/bheisler/criterion.rs) and a [`benchmark-db.sh`](https://github.com/slatedb/slatedb/blob/f88be86d17ac53260d3684edbc8f82811d945b5c/slatedb-bencher/benchmark-db.sh) script. We also run a nightly `benchmark-db.sh` job against a [Tigris](https://www.tigrisdata.com/) object store bucket. (A special thanks to them for donating the bucket to us free of charge.) + +These benchmarks were largely ignored. The output was buried in a [GitHub Actions](https://docs.github.com/actions) summary. It didn't alert when significant regressions occurred, either. If you managed to find it, the output was hard to read. + +Our goal was to make SlateDB's benchmarks easier to find and understand. We also wanted better diagnostic information so developers could visualize SlateDB's behavior under different workloads. + +## Benchmark suite + +We drew inspiration from [ClickBench](https://github.com/ClickHouse/ClickBench), [DuckDB’s benchmark suite](https://duckdb.org/docs/current/dev/benchmark), Lucene’s [nightly benchmarks](https://lucene.apache.org/core/developer.html), and Apache Iggy’s [public benchmark dashboard](https://iggy.apache.org/blogs/2025/02/17/transparent-benchmarks/). [Tectonic: Bridging Synthetic and Real-World Workloads for Key-Value Benchmarking](https://scholarworks.brandeis.edu/esploro/outputs/conferencePresentation/Tectonic-Bridging-Synthetic-and-Real-World-Workloads/9924594037601921) provides a good overview of the space, for those looking to get up to speed. + +Table 1 in the Tectonic paper is particularly relevant. It compares [YCSB](https://github.com/brianfrankcooper/YCSB), [KVBench](https://dl.acm.org/doi/abs/10.1145/3662165.3662765), RocksDB’s [`db_bench`](https://github.com/facebook/rocksdb/wiki/RocksDB-Overview), and Tectonic's own workloads. + + + +Many of the YCSB and `db_bench` workloads cover the behavior users are likely to care about. We chose to adopt a subset of those. If a user needs a more specialized workload, they should run it themselves. The [benchmark repository](https://github.com/slatedb/slatedb-benchmark/) is available for that purpose. + +We also added cost metrics. Throughput and latency matter, but SlateDB exists in part because object storage changes the economics of persistence. Databases should be cheap to run. If it is not, we need to know. + +The suite is made up of three components: + +- A standalone benchmark repository with the suite and its configuration. +- [GitHub Actions](https://github.com/slatedb/slatedb-benchmark/actions) workflows that run it. +- A public [benchmark website](https://benchmark.slatedb.io/) where the results are easier to inspect. + +## Methodology + +We follow YCSB and `db_bench` workloads where possible, but SlateDB does not map perfectly to either. SlateDB is designed around object storage. A cache miss is far more costly, both in terms of money and latency. Misses incur a metered object store API call and a remote network hop. Rather than adjust our configurations to accommodate this, we opted to keep configuration close to SlateDB's defaults. + +The workloads run on [AWS Graviton](https://aws.amazon.com/ec2/graviton/) machines managed by [WarpBuild](https://warpbuild.com) in `us-east-1`. They talk to an [Amazon S3](https://aws.amazon.com/s3/) bucket in the same region and to [Tigris](https://www.tigrisdata.com/) through its `us-east-1` on-ramp. We use a relatively large machine instance ([m8g.2xlarge](https://aws.amazon.com/ec2/instance-types/m8g/)) because compaction is CPU-intensive. + +We configured SlateDB’s cache to hold roughly 10% of the database. The rest lives in object storage. A `db_bench` run of RocksDB keeps the entire database on local disk. This is clearly an apples to oranges comparison. These settings are closer to the way many users configure SlateDB, though. + +Another tradeoff applies to mixed read/write workloads. A read that misses the cache can take tens or hundreds of milliseconds while data comes back from object storage. If one task does both reads and writes, those misses hold back its write throughput. We could separate readers and writers or add more parallelism to report a stronger aggregate throughput number. We kept the configuration closer to RocksDB's workload instead. + +## Findings + +Our first discovery was that routing over the public internet is unpredictable. At one point, we ran in [Hetzner](https://www.hetzner.com/) and expected a nearby path to our Tigris object store bucket. The traffic took a longer route through a different on-ramp, which showed up in latency. We also found local routing behavior that caused tail-latency problems in `us-east-1`. This led to very sawtoothed ingestion charts as SlateDB throttled writes. We eventually worked through these with our cloud provider. + + + +The ingestion benchmark also pushed us toward a simple form of trivial moves. When ingested data has no overlapping key ranges, SlateDB can move it directly from L0 into the next sorted run without rewriting it. RocksDB [supports this](https://github.com/facebook/rocksdb/wiki/Leveled-Compaction#trivial-move), but we hadn't bothered to implement it. Our `sustained-ingest` workload showed it was worth doing. We've opted to disable the setting in the benchmark so we continue to represent baseline performance. We have [a more sophisticated trivial move](https://github.com/slatedb/slatedb/issues/1110) implementation in the works, too. + +Another result came from our new cost performance metrics. We recently added [distributed compaction](https://slatedb.io/blog/compaction-roadmap) support. Compaction can now run on one or more remote machines. This implementation involves periodically checking persisted compaction state in object storage to see if new work is scheduled. + +Most SlateDB deployments keep the writer, compactor, and garbage collector in one process, though. We were still using remote-style object store polling in that scenario. The benchmark showed that an idle database with default settings cost roughly $40 per-month in API polling fees. We made the local coordination path in-memory while preserving the state needed for remote compactors to participate. The idle cost fell to under $5. In practice, you would tune your compactor settings or close idle databases, but $40 per-month is clearly excessive. + +The benchmarks also changed how we think about disk caching. SlateDB has a block cache and an object-store cache. The block cache holds pieces such as data blocks, index blocks, filters, and metadata. The object-store cache works at a larger granularity; its default partition size is 4 MiB. That granularity is expensive when a cache miss occurs. A request for a small key can trigger a 4 MiB fetch from object storage. This appeared as significant latency spikes in our P99 metrics. + + + +We are now planning a [more capable object-store mirror cache](https://github.com/slatedb/slatedb/issues/1980) with read-through, write-through, and write-back modes. The division of labor is clearer, too. If you want to keep the whole database local, an object-level cache makes sense. If you only want partial local caching, the [Foyer hybrid cache](https://github.com/foyer-rs/foyer) is usually a better fit. It already supports eviction and works with much smaller units of data so a cache miss is less costly. [ZeroFS](http://zerofs.net/) has shown us that a [prefetching cache](https://github.com/Barre/ZeroFS/blob/main/zerofs/src/object_store_prefetch.rs) for spatial locality is also possible. + +## More to come + +Performance tuning is a never-ending endeavor. We will continue driving SlateDB's cost and latency down while improving its throughput. We now have a means to evaluate that progress. The results are available at [benchmark.slatedb.io](https://benchmark.slatedb.io/) if we've piqued your curiosity. From 4d7d8b03dc4f3e8096b19d1bff11f1273c893c7c Mon Sep 17 00:00:00 2001 From: Rohan Date: Mon, 17 Aug 2026 12:58:42 -0400 Subject: [PATCH 35/65] [rfc-30 7/N]: move all wal implementation types under the wal module (#2029) --- rfcs/0030-pluggable-wal.md | 23 +- slatedb/src/clone.rs | 2 +- slatedb/src/db.rs | 24 +- slatedb/src/db/builder.rs | 2 +- slatedb/src/db_reader.rs | 12 +- slatedb/src/fence.rs | 6 +- slatedb/src/garbage_collector.rs | 2 +- slatedb/src/lib.rs | 3 +- slatedb/src/tablestore.rs | 2 +- slatedb/src/wal/mod.rs | 6 +- slatedb/src/wal/{ => slatedb}/admin.rs | 2 +- slatedb/src/wal/{ => slatedb}/gc.rs | 0 slatedb/src/wal/slatedb/iterator.rs | 514 ++++++++++++++++++ slatedb/src/wal/slatedb/mod.rs | 9 + slatedb/src/wal/{ => slatedb}/reader.rs | 6 +- .../sst_builder.rs} | 0 .../{wal_buffer.rs => wal/slatedb/writer.rs} | 38 +- slatedb/src/wal/{ => slatedb}/writer_init.rs | 22 +- slatedb/src/wal_replay.rs | 489 +---------------- 19 files changed, 610 insertions(+), 552 deletions(-) rename slatedb/src/wal/{ => slatedb}/admin.rs (99%) rename slatedb/src/wal/{ => slatedb}/gc.rs (100%) create mode 100644 slatedb/src/wal/slatedb/iterator.rs create mode 100644 slatedb/src/wal/slatedb/mod.rs rename slatedb/src/wal/{ => slatedb}/reader.rs (95%) rename slatedb/src/wal/{wal_sst_builder.rs => slatedb/sst_builder.rs} (100%) rename slatedb/src/{wal_buffer.rs => wal/slatedb/writer.rs} (97%) rename slatedb/src/wal/{ => slatedb}/writer_init.rs (91%) diff --git a/rfcs/0030-pluggable-wal.md b/rfcs/0030-pluggable-wal.md index 245b051c26..1183bd4198 100644 --- a/rfcs/0030-pluggable-wal.md +++ b/rfcs/0030-pluggable-wal.md @@ -88,15 +88,15 @@ replays these WAL files into memtables, filtering out any rows with sequence num **Writes** -Once it's recovered persisted writes, the db hands the WAL (`WalBufferManager`) off to the -Batch Writer task. This task serializes all writes and buffers them in `WalBufferManager`, -which periodically flushes the writes to a new WAL file. `WalBufferManager` notifies blocked +Once it's recovered persisted writes, the db hands the WAL (`SlateDbWalWriter`) off to the +Batch Writer task. This task serializes all writes and buffers them in `SlateDbWalWriter`, +which periodically flushes the writes to a new WAL file. `SlateDbWalWriter` notifies blocked write tasks when writes are durably flushed. **Memtable/L0 Flushing** The Batch Writer task adds writes to the memtable once they've been buffered in -`WalBufferManager`. It "freezes" memtables once they cross the memtable size threshold and +`SlateDbWalWriter`. It "freezes" memtables once they cross the memtable size threshold and annotates the frozen memtable with a `replay_after_wal_id` which holds the ID of some WAL File whose writes are fully covered by the memtable (in the current implementation this is the last durably flushed WAL File). The frozen memtables are picked up by a separate Manifest Writer task, @@ -112,7 +112,7 @@ described above depending on what the user requested. **Checkpoints** -When `WalBufferManager` durably persists a WAL File, it notifies the db, which updates +When `SlateDbWalWriter` durably persists a WAL File, it notifies the db, which updates `last_seen_wal_id` in the manifest with the flushed WAL ID. `DbReader` uses this field to determine the range of WAL Files that should be read for a checkpoint. @@ -812,8 +812,8 @@ benchmark tools that instantiate `DbBench` with a db configured to use a custom ## Packaging -The WAL traits and conformance tests will reside in a new crate called `slatedb-wal`. The native -WAL implementation remains in `slatedb`. +The WAL traits and conformance tests will all reside in the `wal` module. The native +WAL implementation will move to a nested module `wal::slatedb`. ## Alternatives @@ -854,6 +854,15 @@ We could have `WalReader`/`WalIterator` iterate over WAL Files which in turn sup iteration (similar to the CDC `WalReader`). I don't really see the benefit of imposing the extra layering. It also forces implementations to map each write batch to a single WAL File. +**Put WAL Trait Definitions in Separate Create** +Initially this RFC proposed putting the WAL trait defs in a separate crate so that implementors +only need to import that crate (vs all of slatedb). We opted not to go this route for a few reasons: +- This requires moving core slatedb types out to either the new wal crate or to slatedb-common. In + particular, we'd need to move all the manifest definitions (`VersionedManifest`, `Manifest`) and + `RowEntry` as these are used by the writer init and reader/iterator, respectively. +- A separate crate isn't that useful. Implementors will almost always have to import slatedb + anyway to run end-to-end tests. And users of custom WALs would be importing slatedb anyway. + ## Open Questions - ~~This RFC proposes an API for streaming new writes via `WalReader`/`WalIterator`. Should this be diff --git a/slatedb/src/clone.rs b/slatedb/src/clone.rs index 852be58907..62116f7340 100644 --- a/slatedb/src/clone.rs +++ b/slatedb/src/clone.rs @@ -605,7 +605,7 @@ mod tests { use crate::proptest_util::{rng, sample}; use crate::test_utils; use crate::utils::IdGenerator; - use crate::wal::admin::SlateDbWalAdmin; + use crate::wal::slatedb::admin::SlateDbWalAdmin; use crate::wal::{WalAdmin, WalError, WalFileRange, WalGc}; use async_trait::async_trait; use bytes::Bytes; diff --git a/slatedb/src/db.rs b/slatedb/src/db.rs index 3ba6f5967b..83b3fda2dc 100644 --- a/slatedb/src/db.rs +++ b/slatedb/src/db.rs @@ -2137,7 +2137,7 @@ impl WriteHandle { } } -/// Wraps [`WalObserver`] and injects a [`crate::wal_buffer::WalStatusListener`] +/// Wraps [`WalObserver`] and injects a [`crate::wal::WalStatusListener`] /// that updates the oracle and manifest, and drives cross-task notifications about wal events /// via a [`tokio::sync::watch`] channel. #[derive(Clone)] @@ -3262,7 +3262,7 @@ mod tests { assert_eq!( lookup_metric( &metrics_recorder, - crate::wal_buffer::stats::WAL_BUFFER_FLUSHES + crate::wal_buffer_stats::WAL_BUFFER_FLUSHES ) .unwrap(), 0 @@ -3283,7 +3283,7 @@ mod tests { assert_eq!( lookup_metric( &metrics_recorder, - crate::wal_buffer::stats::WAL_BUFFER_FLUSHES + crate::wal_buffer_stats::WAL_BUFFER_FLUSHES ) .unwrap(), 1 @@ -3340,7 +3340,7 @@ mod tests { assert_eq!( lookup_metric( &metrics_recorder, - crate::wal_buffer::stats::WAL_BUFFER_FLUSHES + crate::wal_buffer_stats::WAL_BUFFER_FLUSHES ) .unwrap(), 0 @@ -3360,7 +3360,7 @@ mod tests { assert_eq!( lookup_metric( &metrics_recorder, - crate::wal_buffer::stats::WAL_BUFFER_FLUSHES + crate::wal_buffer_stats::WAL_BUFFER_FLUSHES ) .unwrap(), 0 @@ -3404,7 +3404,7 @@ mod tests { assert_eq!( lookup_metric( &metrics_recorder, - crate::wal_buffer::stats::WAL_BUFFER_FLUSHES + crate::wal_buffer_stats::WAL_BUFFER_FLUSHES ) .unwrap(), 0 @@ -3418,7 +3418,7 @@ mod tests { assert_eq!( lookup_metric( &metrics_recorder, - crate::wal_buffer::stats::WAL_BUFFER_FLUSHES + crate::wal_buffer_stats::WAL_BUFFER_FLUSHES ) .unwrap(), 1 @@ -3565,7 +3565,7 @@ mod tests { .unwrap(); assert_eq!( - lookup_metric(&metrics_recorder, crate::wal_buffer::stats::WAL_FLUSH_BYTES) + lookup_metric(&metrics_recorder, crate::wal_buffer_stats::WAL_FLUSH_BYTES,) .unwrap_or(0), 0 ); @@ -3726,7 +3726,7 @@ mod tests { .await .unwrap(); assert_eq!( - lookup_metric(&metrics_recorder, crate::wal_buffer::stats::WAL_FLUSH_BYTES) + lookup_metric(&metrics_recorder, crate::wal_buffer_stats::WAL_FLUSH_BYTES,) .unwrap_or(0), 0, ); @@ -3734,7 +3734,7 @@ mod tests { db.flush().await.unwrap(); let wal_bytes = - lookup_metric(&metrics_recorder, crate::wal_buffer::stats::WAL_FLUSH_BYTES).unwrap(); + lookup_metric(&metrics_recorder, crate::wal_buffer_stats::WAL_FLUSH_BYTES).unwrap(); let memtable_bytes = lookup_metric(&metrics_recorder, crate::db_stats::MEMTABLE_WRITE_BYTES).unwrap(); // WAL SST framing/footer makes the encoded payload at least as large as @@ -3760,7 +3760,7 @@ mod tests { tokio::time::timeout( Duration::from_secs(1), db.task_executor - .join_task(crate::wal_buffer::WAL_BUFFER_TASK_NAME), + .join_task(crate::wal::slatedb::writer::WAL_BUFFER_TASK_NAME), ) .await .expect("native WAL task should not run when the WAL is disabled") @@ -10068,7 +10068,7 @@ mod tests { // then: let estimated = lookup_metric( &metrics_recorder, - crate::wal_buffer::stats::WAL_BUFFER_ESTIMATED_BYTES, + crate::wal_buffer_stats::WAL_BUFFER_ESTIMATED_BYTES, ); assert!( estimated.is_some_and(|v| v > 0), diff --git a/slatedb/src/db/builder.rs b/slatedb/src/db/builder.rs index aecfbef28d..c59c82ce99 100644 --- a/slatedb/src/db/builder.rs +++ b/slatedb/src/db/builder.rs @@ -162,7 +162,7 @@ use crate::tablestore::{TableStore, TableStoreKind}; use crate::utils::SafeSender; use crate::utils::WatchableOnceCell; use crate::wal; -use crate::wal::admin::SlateDbWalAdmin; +use crate::wal::slatedb::admin::SlateDbWalAdmin; use crate::wal::wal_disabled::DisabledWalObserver; use crate::wal::{WalAdmin, WalGc, WalObserver}; use slatedb_common::clock::DefaultSystemClock; diff --git a/slatedb/src/db_reader.rs b/slatedb/src/db_reader.rs index abe16e28a6..b7decbb385 100644 --- a/slatedb/src/db_reader.rs +++ b/slatedb/src/db_reader.rs @@ -183,9 +183,9 @@ impl DbReaderInner { mut manifest: StoredManifest, ) -> Result { let wal_reader = wal_reader.unwrap_or_else(|| { - Arc::new(crate::wal::reader::SlateDbWalReader::new(Arc::clone( - &table_store, - ))) + Arc::new(crate::wal::slatedb::reader::SlateDbWalReader::new( + Arc::clone(&table_store), + )) }); let checkpoint = Self::get_or_create_checkpoint(&mut manifest, mode, &options, rand.clone()).await?; @@ -3079,8 +3079,10 @@ mod tests { } } - fn native_wal_reader(table_store: &Arc) -> crate::wal::reader::SlateDbWalReader { - crate::wal::reader::SlateDbWalReader::new(Arc::clone(table_store)) + fn native_wal_reader( + table_store: &Arc, + ) -> crate::wal::slatedb::reader::SlateDbWalReader { + crate::wal::slatedb::reader::SlateDbWalReader::new(Arc::clone(table_store)) } fn immutable_memtable( diff --git a/slatedb/src/fence.rs b/slatedb/src/fence.rs index 182d7ef752..5e558061df 100644 --- a/slatedb/src/fence.rs +++ b/slatedb/src/fence.rs @@ -3,7 +3,7 @@ use crate::error::SlateDBError; use crate::manifest::store::{FenceableManifest, StoredManifest}; use crate::tablestore::TableStore; use crate::utils::WatchableOnceCellReader; -use crate::wal::writer_init::{WalWriterInit, WalWriterInitOptions}; +use crate::wal::slatedb::writer_init::{SlateDbWalWriterInit, SlateDbWalWriterInitOptions}; use crate::wal::{WalIterator, WalWriter, WriterInit}; use crate::Settings; use fail_parallel::{fail_point_send, FailPointTx}; @@ -15,7 +15,7 @@ use std::time::Duration; pub(crate) struct WriterFencer { closed_result_reader: WatchableOnceCellReader>, recorder: MetricsRecorderHelper, - wal_writer_init_options: WalWriterInitOptions, + wal_writer_init_options: SlateDbWalWriterInitOptions, table_store: Arc, manifest_update_timeout: Duration, system_clock: Arc, @@ -91,7 +91,7 @@ impl WriterFencer { let wal_writer_init = match self.wal_writer_init.take() { Some(wal_writer_init) => wal_writer_init, None => Box::new( - WalWriterInit::load( + SlateDbWalWriterInit::load( self.closed_result_reader.clone(), self.recorder.clone(), self.table_store.clone(), diff --git a/slatedb/src/garbage_collector.rs b/slatedb/src/garbage_collector.rs index 674840ad39..4ab9d4f7a5 100644 --- a/slatedb/src/garbage_collector.rs +++ b/slatedb/src/garbage_collector.rs @@ -24,7 +24,7 @@ use crate::manifest::store::{ManifestStore, StoredManifest}; use crate::manifest::Manifest; use crate::tablestore::TableStore; use crate::utils::WatchableOnceCell; -use crate::wal::gc::{SlateDbWalGc, WalGcMode}; +use crate::wal::slatedb::gc::{SlateDbWalGc, WalGcMode}; use async_trait::async_trait; use chrono::{DateTime, Utc}; use compacted_gc::CompactedGcTask; diff --git a/slatedb/src/lib.rs b/slatedb/src/lib.rs index 3bfd81214e..c14fda82dc 100644 --- a/slatedb/src/lib.rs +++ b/slatedb/src/lib.rs @@ -77,7 +77,7 @@ pub use sst_stats::{BlockStats, SstStats}; pub use transaction_manager::IsolationLevel; pub use types::KeyValue; pub use types::{RowEntry, ValueDeletable}; -pub use wal_buffer::stats as wal_buffer_stats; +pub use wal::slatedb::writer::stats as wal_buffer_stats; pub use wal_reader::{WalFile, WalFileIterator, WalReader}; pub mod admin; @@ -174,7 +174,6 @@ mod types; mod utils; mod fence; -mod wal_buffer; mod wal_reader; mod wal_replay; diff --git a/slatedb/src/tablestore.rs b/slatedb/src/tablestore.rs index 5a5b16e32b..e496824743 100644 --- a/slatedb/src/tablestore.rs +++ b/slatedb/src/tablestore.rs @@ -33,7 +33,7 @@ use crate::paths::PathResolver; use crate::sst_builder::EncodedSsTableBuilder; use crate::sst_stats::SstStats; use crate::types::RowEntry; -use crate::wal::wal_sst_builder::EncodedWalSsTableBuilder; +use crate::wal::slatedb::sst_builder::EncodedWalSsTableBuilder; pub(crate) struct TableStore { object_stores: ObjectStores, diff --git a/slatedb/src/wal/mod.rs b/slatedb/src/wal/mod.rs index fc11c2914b..15546cdf5a 100644 --- a/slatedb/src/wal/mod.rs +++ b/slatedb/src/wal/mod.rs @@ -10,14 +10,10 @@ use std::ops::{Bound, Range}; use std::sync::Arc; use std::time::Duration; -pub(crate) mod admin; -pub(crate) mod gc; -pub(crate) mod reader; +pub(crate) mod slatedb; #[cfg(test)] pub(crate) mod test_utils; pub(crate) mod wal_disabled; -pub(crate) mod wal_sst_builder; -pub(crate) mod writer_init; /// A range of WAL File IDs #[derive(Clone, Debug, Eq, PartialEq)] diff --git a/slatedb/src/wal/admin.rs b/slatedb/src/wal/slatedb/admin.rs similarity index 99% rename from slatedb/src/wal/admin.rs rename to slatedb/src/wal/slatedb/admin.rs index 855fa634b6..7d586f144a 100644 --- a/slatedb/src/wal/admin.rs +++ b/slatedb/src/wal/slatedb/admin.rs @@ -5,7 +5,7 @@ use crate::garbage_collector::stats::GcStats; use crate::object_stores::ObjectStores; use crate::paths::PathResolver; use crate::tablestore::{TableStore, TableStoreKind}; -use crate::wal::gc::{SlateDbWalGc, WalGcMode}; +use crate::wal::slatedb::gc::{SlateDbWalGc, WalGcMode}; use crate::wal::{WalAdmin, WalError, WalGc}; use crate::VersionedManifest; use async_trait::async_trait; diff --git a/slatedb/src/wal/gc.rs b/slatedb/src/wal/slatedb/gc.rs similarity index 100% rename from slatedb/src/wal/gc.rs rename to slatedb/src/wal/slatedb/gc.rs diff --git a/slatedb/src/wal/slatedb/iterator.rs b/slatedb/src/wal/slatedb/iterator.rs new file mode 100644 index 0000000000..7baad7fd2b --- /dev/null +++ b/slatedb/src/wal/slatedb/iterator.rs @@ -0,0 +1,514 @@ +use std::collections::VecDeque; +use std::ops::Range; +use std::sync::Arc; + +use async_trait::async_trait; +use log::error; +use tokio::task; +use tokio::task::JoinHandle; + +use crate::db_state::SsTableId; +use crate::error::SlateDBError; +use crate::iter::{EmptyIterator, RowEntryIterator}; +use crate::manifest::SsTableView; +use crate::sst_iter::{SstIterator, SstIteratorOptions}; +use crate::tablestore::TableStore; +use crate::utils::panic_string; +use crate::wal::{WalError, WalIterator as WalIteratorTrait, WalRows}; +use crate::RowEntry; + +pub(crate) struct SlateDbWalIteratorOptions { + /// The number of SSTs to preload while replaying + pub(crate) sst_batch_size: usize, + + /// Options to pass through to underlying SST iterators + pub(crate) sst_iter_options: SstIteratorOptions, +} + +impl Default for SlateDbWalIteratorOptions { + fn default() -> Self { + Self { + sst_batch_size: 4, + sst_iter_options: SstIteratorOptions::default(), + } + } +} + +struct WalRowsCollector { + wal_id: u64, + iter: Box, + rows: Vec, + drained: bool, +} + +impl WalRowsCollector { + fn new(wal_id: u64, iter: Box) -> Self { + Self { + wal_id, + iter, + rows: vec![], + drained: false, + } + } + + async fn collect(&mut self) -> Result<(), WalError> { + loop { + match self.iter.next().await { + Ok(Some(row)) => self.rows.push(row), + Ok(None) => { + self.drained = true; + break Ok(()); + } + Err(err) if err.has_object_store_not_found() => { + break Err(WalError::WalTruncated(self.wal_id)); + } + Err(err) => { + break Err(err.into()); + } + } + } + } +} + +impl From for WalRows { + fn from(reader: WalRowsCollector) -> Self { + assert!(reader.drained); + WalRows { + last_consumed_wal_file_id: reader.wal_id, + rows: reader.rows, + } + } +} + +struct CurrentWalFile { + initialized: bool, + collector: Option, +} + +impl CurrentWalFile { + fn initial() -> Self { + Self { + initialized: false, + collector: None, + } + } + + fn initialized(&self) -> bool { + self.initialized + } + + async fn collect(&mut self) -> Result, WalError> { + assert!(self.initialized); + let Some(collector) = &mut self.collector else { + return Ok(None); + }; + collector.collect().await?; + let collector = self.collector.take().expect("unreachable"); + self.initialized = false; + Ok(Some(collector.into())) + } + + fn advance(&mut self, collector: WalRowsCollector) { + assert!(!self.initialized); + self.initialized = true; + self.collector = Some(collector); + } + + fn finish(&mut self) { + self.initialized = true; + self.collector = None; + } +} + +/// Iterates over the writes in a range of WAL files, preloading up to +/// `sst_batch_size` WAL SSTs concurrently. Returns the rows of one WAL file per +/// [`WalRows`], and verifies that files carry strictly increasing seq +/// ranges — the ordering callers rely on to split and tag memtables safely. +/// +/// Preloading only opens each WAL SST (footer, index, and any eagerly fetched +/// blocks); a file's rows are read out only when it is returned from +/// [`Self::next`], so at most one file's rows are materialized at a time. +pub(crate) struct SlateDbWalIterator { + options: SlateDbWalIteratorOptions, + /// Range of WAL IDs to iterate over + wal_id_range: Range, + table_store: Arc, + next_files: VecDeque>>, + next_wal_id: u64, + /// The greatest seq returned so far, used to verify that WAL files arrive + /// with strictly increasing seq ranges. + last_seq: Option, + /// Set once iteration has ended, either because the range was exhausted or + /// because an error was returned. + terminal_result: Option, WalError>>, + current_file: CurrentWalFile, +} + +impl SlateDbWalIterator { + pub(crate) fn range( + wal_id_range: Range, + options: SlateDbWalIteratorOptions, + table_store: Arc, + ) -> Result { + if options.sst_batch_size < 1 { + return Err(SlateDBError::InvalidSSTBatchSize(options.sst_batch_size)); + } + + let next_wal_id = wal_id_range.start; + Ok(Self { + options, + wal_id_range, + table_store, + next_files: VecDeque::new(), + next_wal_id, + last_seq: None, + terminal_result: None, + current_file: CurrentWalFile::initial(), + }) + } + + fn maybe_spawn_open(&mut self) -> bool { + if !self.wal_id_range.contains(&self.next_wal_id) + || self.next_files.len() >= self.options.sst_batch_size + { + return false; + } + + let next_wal_id = self.next_wal_id; + self.next_wal_id += 1; + + async fn try_open_file_iter( + wal_id: u64, + sst_iter_options: SstIteratorOptions, + table_store: Arc, + ) -> Result { + let sst = match table_store.open_sst(&SsTableId::Wal(wal_id)).await { + Ok(sst) => sst, + Err(SlateDBError::EmptySSTable) => { + // Zero-byte WAL files are fence markers; replay them as empty WALs + // so the last replayed WAL ID still advances past the marker. + return Ok(WalRowsCollector::new( + wal_id, + Box::new(EmptyIterator::new()), + )); + } + Err(err) => return Err(err), + }; + let iter = SstIterator::new_owned_initialized( + .., + SsTableView::identity(sst), + Arc::clone(&table_store), + sst_iter_options, + ) + .await?; + // An unbounded, unfiltered scan over a WAL SST always yields an + // iterator. `None` means the file cannot be read, and replay must + // fail rather than silently end early and drop the remaining WALs. + let Some(iter) = iter else { + error!( + "could not construct row iterator over WAL SST. [wal_id={}]", + wal_id + ); + return Err(SlateDBError::InvalidDBState); + }; + Ok(WalRowsCollector::new(wal_id, Box::new(iter))) + } + + async fn open_file_iter( + wal_id: u64, + sst_iter_options: SstIteratorOptions, + table_store: Arc, + ) -> Result { + match try_open_file_iter(wal_id, sst_iter_options, table_store).await { + Ok(iter) => Ok(iter), + Err(err) if err.has_object_store_not_found() => Err(WalError::WalTruncated(wal_id)), + Err(err) => Err(err.into()), + } + } + + let handle = task::spawn(open_file_iter( + next_wal_id, + self.options.sst_iter_options.clone(), + Arc::clone(&self.table_store), + )); + self.next_files.push_back(handle); + true + } + + /// Await the next preloaded WAL file and return an iterator over its rows. + /// Returns `None` when there are no more files to read. + async fn load_next_file(&mut self) -> Result<(), WalError> { + if self.current_file.initialized() { + return Ok(()); + } + // await a mutable ref to the task so that next remains cancel-safe + // see https://docs.rs/tokio/latest/tokio/task/struct.JoinHandle.html#cancel-safety + let Some(join_handle) = self.next_files.front_mut() else { + self.current_file.finish(); + return Ok(()); + }; + let result = join_handle.await; + self.next_files.pop_front(); + match result { + Ok(result) => { + self.current_file.advance(result?); + Ok(()) + } + Err(join_err) => { + let task_name = format!("wal_replay[{:?}]", self.wal_id_range); + let msg = if let Ok(panic_err) = join_err.try_into_panic() { + format!( + "wal_replay task panicked unexpectedly. [task_name={}, panic={}]", + task_name, + panic_string(&panic_err), + ) + } else { + format!("wal_replay task cancelled. [task_name={}]", task_name) + }; + error!("{}", msg); + let error = Arc::from(Box::::from(msg)); + Err(WalError::InternalError(error)) + } + } + } + + fn terminate( + &mut self, + result: Result, WalError>, + ) -> Result, WalError> { + self.terminal_result = Some(result.clone()); + for task in self.next_files.drain(..) { + task.abort(); + } + result + } +} + +#[async_trait] +impl WalIteratorTrait for SlateDbWalIterator { + /// Get the next set of writes from the WAL files in the range. Each returned + /// [`WalRows`] holds the rows of one WAL file; a WAL file with no rows + /// yields a batch with empty `rows`. Returns `None` once all WAL files in the + /// range have been read. It is an error if a WAL file in the range is not + /// present. Errors are returned only on calls that return no batch, so rows + /// read from earlier WAL files are never dropped with a later file's error. + async fn next(&mut self) -> Result, WalError> { + if let Some(result) = self.terminal_result.clone() { + return result; + } + + while self.maybe_spawn_open() {} + if let Err(err) = self.load_next_file().await { + return self.terminate(Err(err)); + } + match self.current_file.collect().await { + Err(err) => self.terminate(Err(err)), + Ok(None) => self.terminate(Ok(None)), + Ok(Some(rows)) => { + // Verify that WAL files carry strictly increasing seq ranges. Replay + // relies on this ordering to split and tag memtables safely: a commit seq + // spanning two WAL files, or files with overlapping seq ranges, would + // break recovery's (wal_id, seq) watermark filtering. + if let Some(min_seq) = rows.rows.iter().map(|row| row.seq).min() { + if let Some(last_seq) = self.last_seq { + if min_seq <= last_seq { + let msg = format!( + "WAL replay saw out-of-order seqs across WAL files. \ + [wal_id={}, min_seq={}, last_seq={}]", + rows.last_consumed_wal_file_id, min_seq, last_seq, + ); + error!("{}", &msg); + let error = + Arc::from(Box::::from(msg)); + return self.terminate(Err(WalError::InternalError(error))); + } + } + let max_seq = rows + .rows + .iter() + .map(|row| row.seq) + .max() + .expect("non-empty rows have a max seq"); + self.last_seq = Some(max_seq); + } + Ok(Some(rows)) + } + } + } +} + +#[cfg(test)] +mod tests { + use std::collections::{BTreeMap, BTreeSet}; + use std::sync::Arc; + + use bytes::Bytes; + use object_store::memory::InMemory; + use object_store::path::Path; + use object_store::ObjectStore; + + use super::{SlateDbWalIterator, SlateDbWalIteratorOptions}; + use crate::block_cache_policy::BlockCachePolicy; + use crate::db_state::SsTableId; + use crate::format::sst::SsTableFormat; + use crate::object_stores::ObjectStores; + use crate::tablestore::{TableStore, TableStoreKind}; + use crate::types::RowEntry; + use crate::wal::{WalError, WalIterator as _}; + + #[tokio::test] + async fn should_repeat_terminal_error_for_wal_iterator() { + let table_store = test_table_store(); + let mut wal_iter = SlateDbWalIterator::range( + 1..2, + SlateDbWalIteratorOptions::default(), + Arc::clone(&table_store), + ) + .unwrap(); + + assert!(matches!( + wal_iter.next().await, + Err(WalError::WalTruncated(1)) + )); + assert!(matches!( + wal_iter.next().await, + Err(WalError::WalTruncated(1)) + )); + } + + #[tokio::test] + async fn should_repeat_terminal_none_for_wal_iterator() { + let table_store = test_table_store(); + let mut wal_iter = SlateDbWalIterator::range( + 1..1, + SlateDbWalIteratorOptions::default(), + Arc::clone(&table_store), + ) + .unwrap(); + + assert!(wal_iter.next().await.unwrap().is_none()); + assert!(wal_iter.next().await.unwrap().is_none()); + } + + #[tokio::test] + async fn should_return_atomic_wal_rows_in_increasing_seq_order() { + let table_store = test_table_store(); + // Each file contains out-of-order rows and a sequence that appears twice. + // The iterator must keep both rows for a sequence in one batch, while the + // sequence range of the second batch must follow the first. + let wal_entries = [ + vec![ + RowEntry::new_value(b"key_001", &[b'x'; 128], 2), + RowEntry::new_value(b"key_002", &[b'x'; 128], 1), + RowEntry::new_value(b"key_003", &[b'x'; 128], 2), + ], + vec![ + RowEntry::new_value(b"key_004", &[b'x'; 128], 4), + RowEntry::new_value(b"key_005", &[b'x'; 128], 3), + RowEntry::new_value(b"key_006", &[b'x'; 128], 4), + ], + ]; + let mut expected_rows = BTreeMap::new(); + let mut expected_rows_by_seq = BTreeMap::>::new(); + let wal_file_count = wal_entries.len() as u64; + for (file_index, entries) in wal_entries.iter().enumerate() { + let wal_id = file_index as u64 + 1; + for row in entries { + expected_rows.insert(row.key.clone(), (row.clone(), wal_id)); + expected_rows_by_seq + .entry(row.seq) + .or_default() + .insert(row.key.clone()); + } + } + for (index, entries) in wal_entries.into_iter().enumerate() { + let mut builder = table_store.wal_table_builder(); + for entry in entries { + builder.add(entry).await.unwrap(); + } + let encoded_sst = builder.build().await.unwrap(); + table_store + .write_sst(&SsTableId::Wal(index as u64 + 1), &encoded_sst) + .await + .unwrap(); + } + let mut wal_iter = SlateDbWalIterator::range( + 1..(wal_file_count + 1), + SlateDbWalIteratorOptions::default(), + Arc::clone(&table_store), + ) + .unwrap(); + + let mut returned_rows = BTreeMap::new(); + let mut previous_max_seq = None; + let mut last_consumed_wal_file_id = 0; + while let Some(batch) = wal_iter.next().await.unwrap() { + let batch_min_seq = batch.rows.iter().map(|r| r.seq).min().unwrap(); + let batch_max_seq = batch.rows.iter().map(|r| r.seq).max().unwrap(); + if let Some(previous_max_seq) = previous_max_seq { + assert!( + batch_min_seq > previous_max_seq, + "consecutive WAL batches have overlapping sequence ranges" + ); + } + previous_max_seq = Some(batch_max_seq); + + let mut batch_rows_by_seq = BTreeMap::>::new(); + for row in &batch.rows { + assert!( + returned_rows.insert(row.key.clone(), row.clone()).is_none(), + "row was returned more than once: {:?}", + row.key + ); + batch_rows_by_seq + .entry(row.seq) + .or_default() + .insert(row.key.clone()); + } + for (seq, batch_rows) in batch_rows_by_seq { + assert_eq!( + expected_rows_by_seq.get(&seq), + Some(&batch_rows), + "rows for seq {seq} were split across WAL batches" + ); + } + + assert!( + batch.last_consumed_wal_file_id >= last_consumed_wal_file_id, + "consumed WAL file watermark moved backwards" + ); + assert!(batch.last_consumed_wal_file_id <= wal_file_count); + for wal_id in 1..=batch.last_consumed_wal_file_id { + let file_fully_returned = + expected_rows.iter().all(|(key, (_, expected_wal_id))| { + *expected_wal_id != wal_id || returned_rows.contains_key(key) + }); + assert!( + file_fully_returned, + "WAL file {wal_id} was marked consumed before all its rows were returned" + ); + } + last_consumed_wal_file_id = batch.last_consumed_wal_file_id; + } + + let expected_returned_rows = expected_rows + .into_iter() + .map(|(key, (row, _wal_id))| (key, row)) + .collect(); + assert_eq!(returned_rows, expected_returned_rows); + assert_eq!(last_consumed_wal_file_id, wal_file_count); + } + + fn test_table_store() -> Arc { + let object_store: Arc = Arc::new(InMemory::new()); + let path = Path::from("/tmp/test_kv_store"); + Arc::new(TableStore::new( + ObjectStores::new(object_store.clone(), None), + SsTableFormat::default(), + path, + None, + TableStoreKind::Main, + BlockCachePolicy::default(), + )) + } +} diff --git a/slatedb/src/wal/slatedb/mod.rs b/slatedb/src/wal/slatedb/mod.rs new file mode 100644 index 0000000000..b0e314d394 --- /dev/null +++ b/slatedb/src/wal/slatedb/mod.rs @@ -0,0 +1,9 @@ +//! SlateDB's native object-store-backed WAL implementation. + +pub(crate) mod admin; +pub(crate) mod gc; +pub(crate) mod iterator; +pub(crate) mod reader; +pub(crate) mod sst_builder; +pub(crate) mod writer; +pub(crate) mod writer_init; diff --git a/slatedb/src/wal/reader.rs b/slatedb/src/wal/slatedb/reader.rs similarity index 95% rename from slatedb/src/wal/reader.rs rename to slatedb/src/wal/slatedb/reader.rs index be0dd54460..77a89d882f 100644 --- a/slatedb/src/wal/reader.rs +++ b/slatedb/src/wal/slatedb/reader.rs @@ -5,8 +5,8 @@ use async_trait::async_trait; use crate::iter::IterationOrder; use crate::sst_iter::SstIteratorOptions; use crate::tablestore::TableStore; +use crate::wal::slatedb::iterator::{SlateDbWalIterator, SlateDbWalIteratorOptions}; use crate::wal::{WalError, WalFileRange, WalIterator, WalReader}; -use crate::wal_replay::{WalIterator as WalReplayIterator, WalIteratorOptions}; pub(crate) struct SlateDbWalReader { table_store: Arc, @@ -30,9 +30,9 @@ impl WalReader for SlateDbWalReader { "native WAL reader requires an included start and excluded end", ))) })?; - let iterator = WalReplayIterator::range( + let iterator = SlateDbWalIterator::range( wal_id_range, - WalIteratorOptions { + SlateDbWalIteratorOptions { sst_batch_size: 4, sst_iter_options: SstIteratorOptions { max_fetch_tasks: 1, diff --git a/slatedb/src/wal/wal_sst_builder.rs b/slatedb/src/wal/slatedb/sst_builder.rs similarity index 100% rename from slatedb/src/wal/wal_sst_builder.rs rename to slatedb/src/wal/slatedb/sst_builder.rs diff --git a/slatedb/src/wal_buffer.rs b/slatedb/src/wal/slatedb/writer.rs similarity index 97% rename from slatedb/src/wal_buffer.rs rename to slatedb/src/wal/slatedb/writer.rs index b0fe873b39..4cebb1216c 100644 --- a/slatedb/src/wal_buffer.rs +++ b/slatedb/src/wal/slatedb/writer.rs @@ -4,6 +4,7 @@ use std::sync::atomic::{AtomicU64, Ordering}; use std::sync::Arc; use std::time::Duration; +use self::stats::WalBufferStats; use crate::db_state::SsTableId; use crate::dispatcher::{MessageHandler, MessageHandlerExecutor, MessageTickerDef}; use crate::error::SlateDBError; @@ -13,7 +14,6 @@ use crate::utils::SafeSender; use crate::utils::{format_bytes_si, WatchableOnceCellReader}; use crate::wal; use crate::wal::{FlushResultFuture, WalError, WalEvent, WalStatus, WalWriter}; -use crate::wal_buffer_stats::WalBufferStats; use async_trait::async_trait; use futures::{stream::BoxStream, FutureExt, StreamExt}; use log::{error, trace, warn}; @@ -23,7 +23,7 @@ use tracing::instrument; pub(crate) const WAL_BUFFER_TASK_NAME: &str = "wal_writer"; -/// [`WalBufferManager`] buffers write operations in memory before flushing them to persistent storage. +/// [`SlateDbWalWriter`] buffers write operations in memory before flushing them to persistent storage. /// The flush operation only targets Remote storage right now, later we can add an option to flush to local /// storage. /// @@ -36,7 +36,7 @@ pub(crate) const WAL_BUFFER_TASK_NAME: &str = "wal_writer"; /// - `max_wal_flushes_before_l0_flush`: Requests a memtable flush when this many WAL files have /// been flushed since the latest memtable replay point /// -/// For strict durability requirements on synchronous writes, use [`WalBufferManager::flush()`] to explicitly +/// For strict durability requirements on synchronous writes, use [`SlateDbWalWriter::flush()`] to explicitly /// trigger a flush operation and await the result. This will flush ALL the in memory WALs (including the /// current WAL) to remote storage. /// @@ -48,8 +48,8 @@ pub(crate) const WAL_BUFFER_TASK_NAME: &str = "wal_writer"; /// guaranteed to be written atomically to the same WAL file. /// - Fatal errors during flush operations are stored internally and propagated to all subsequent /// operations. The manager becomes unusable after encountering a fatal error. -pub(crate) struct WalBufferManager { - inner: Arc>, +pub(crate) struct SlateDbWalWriter { + inner: Arc>, stats: Arc, table_store: Arc, max_wal_bytes_size: usize, @@ -62,7 +62,7 @@ pub(crate) struct WalBufferManager { task_executor: Arc, } -struct WalBufferManagerInner { +struct SlateDbWalWriterInner { current_wal: WalBuffer, /// When the current WAL is ready to be flushed, it'll be moved to the `immutable_wals`. /// The flusher will try flush all the immutable wals to remote storage. @@ -107,7 +107,7 @@ struct WalBufferIterator { entries: std::vec::IntoIter, } -impl WalBufferManager { +impl SlateDbWalWriter { pub(crate) async fn start_new( closed_result_reader: WatchableOnceCellReader>, recorder: &MetricsRecorderHelper, @@ -121,7 +121,7 @@ impl WalBufferManager { let current_wal = WalBuffer::new(); let immutable_wals = VecDeque::new(); let (flush_tx, flush_rx) = SafeSender::unbounded_channel(closed_result_reader); - let inner = WalBufferManagerInner { + let inner = SlateDbWalWriterInner { current_wal, immutable_wals, flush_epoch: 1, @@ -203,7 +203,7 @@ impl WalBufferManager { } #[async_trait] -impl WalWriter for WalBufferManager { +impl WalWriter for SlateDbWalWriter { fn status(&self) -> Result { self.inner.read().status(&self.table_store) } @@ -224,7 +224,7 @@ impl WalWriter for WalBufferManager { } fn observer(&self) -> Box { - Box::new(WalObserver { + Box::new(SlateDbWalObserver { inner: self.inner.clone(), table_store: self.table_store.clone(), }) @@ -256,7 +256,7 @@ impl WalWriter for WalBufferManager { } } -impl WalBufferManagerInner { +impl SlateDbWalWriterInner { fn check_exited(&self) -> Result<(), WalError> { match self.flush_task_exited_reason.as_ref() { Some(err) => Err(err.clone()), @@ -466,7 +466,7 @@ impl Debug for WalFlushWork { struct WalFlushHandler { max_flush_interval: Option, - inner: Arc>, + inner: Arc>, table_store: Arc, stats: Arc, listener: Option, @@ -601,12 +601,12 @@ impl MessageHandler for WalFlushHandler { /// Interface for getting information about the current state of the Wal #[derive(Clone)] -struct WalObserver { - inner: Arc>, +struct SlateDbWalObserver { + inner: Arc>, table_store: Arc, } -impl wal::WalObserver for WalObserver { +impl wal::WalObserver for SlateDbWalObserver { /// Gets information about the Wal buffer's current state fn status(&self) -> Result { self.inner.read().status(self.table_store.as_ref()) @@ -801,7 +801,7 @@ mod tests { } async fn setup_wal_buffer() -> ( - WalBufferManager, + SlateDbWalWriter, Arc, Arc, Arc, @@ -812,7 +812,7 @@ mod tests { async fn setup_wal_buffer_with_flush_interval( flush_interval: Duration, ) -> ( - WalBufferManager, + SlateDbWalWriter, Arc, Arc, Arc, @@ -824,7 +824,7 @@ mod tests { flush_interval: Duration, listener: wal::WalStatusListener, ) -> ( - WalBufferManager, + SlateDbWalWriter, Arc, Arc, Arc, @@ -847,7 +847,7 @@ mod tests { status_manager.clone(), system_clock.clone(), )); - let wal_buffer = WalBufferManager::start_new( + let wal_buffer = SlateDbWalWriter::start_new( status_manager.result_reader(), &helper, 0, // recent_flushed_wal_id diff --git a/slatedb/src/wal/writer_init.rs b/slatedb/src/wal/slatedb/writer_init.rs similarity index 91% rename from slatedb/src/wal/writer_init.rs rename to slatedb/src/wal/slatedb/writer_init.rs index ddf035deea..ec1a6be609 100644 --- a/slatedb/src/wal/writer_init.rs +++ b/slatedb/src/wal/slatedb/writer_init.rs @@ -5,9 +5,9 @@ use crate::manifest::Manifest; use crate::sst_iter::SstIteratorOptions; use crate::tablestore::TableStore; use crate::utils::WatchableOnceCellReader; +use crate::wal::slatedb::iterator::{SlateDbWalIterator, SlateDbWalIteratorOptions}; +use crate::wal::slatedb::writer::SlateDbWalWriter; use crate::wal::{WalError, WriterInitResult, WriterManifest}; -use crate::wal_buffer::WalBufferManager; -use crate::wal_replay::{WalIterator, WalIteratorOptions}; use crate::{wal, Settings}; use async_trait::async_trait; use fail_parallel::{fail_point_send, FailPointTx}; @@ -16,13 +16,13 @@ use std::sync::Arc; use std::time::Duration; #[derive(Clone, Copy)] -pub(crate) struct WalWriterInitOptions { +pub(crate) struct SlateDbWalWriterInitOptions { max_wal_bytes_size: usize, max_wal_flushes_before_l0_flush: u64, max_flush_interval: Option, } -impl From<&Settings> for WalWriterInitOptions { +impl From<&Settings> for SlateDbWalWriterInitOptions { fn from(settings: &Settings) -> Self { Self { max_wal_bytes_size: settings.l0_sst_size_bytes, @@ -32,7 +32,7 @@ impl From<&Settings> for WalWriterInitOptions { } } -pub(crate) struct WalWriterInit { +pub(crate) struct SlateDbWalWriterInit { closed_result_reader: WatchableOnceCellReader>, recorder: MetricsRecorderHelper, table_store: Arc, @@ -45,12 +45,12 @@ pub(crate) struct WalWriterInit { fp_tx: FailPointTx, } -impl WalWriterInit { +impl SlateDbWalWriterInit { pub(crate) async fn load( closed_result_reader: WatchableOnceCellReader>, recorder: MetricsRecorderHelper, table_store: Arc, - options: WalWriterInitOptions, + options: SlateDbWalWriterInitOptions, manifest: &Manifest, task_executor: Arc, fp_tx: FailPointTx, @@ -74,7 +74,7 @@ impl WalWriterInit { } #[async_trait] -impl wal::WriterInit for WalWriterInit { +impl wal::WriterInit for SlateDbWalWriterInit { async fn fence_and_init( &self, writer_manifest: &mut WriterManifest, @@ -116,9 +116,9 @@ impl wal::WriterInit for WalWriterInit { // older writers would have failed with a stale epoch let replay_after_wal_id = manifest.core().replay_after_wal_id; assert!(empty_wal_id > replay_after_wal_id); - let replay_iterator = WalIterator::range( + let replay_iterator = SlateDbWalIterator::range( replay_after_wal_id + 1..empty_wal_id + 1, - WalIteratorOptions { + SlateDbWalIteratorOptions { sst_batch_size: 4, sst_iter_options: SstIteratorOptions { max_fetch_tasks: 1, @@ -133,7 +133,7 @@ impl wal::WriterInit for WalWriterInit { }, self.table_store.clone(), )?; - let wal_writer = WalBufferManager::start_new( + let wal_writer = SlateDbWalWriter::start_new( self.closed_result_reader.clone(), &self.recorder, empty_wal_id, diff --git a/slatedb/src/wal_replay.rs b/slatedb/src/wal_replay.rs index e286e6f14d..afd98618de 100644 --- a/slatedb/src/wal_replay.rs +++ b/slatedb/src/wal_replay.rs @@ -1,38 +1,13 @@ -use crate::db_state::SsTableId; use crate::error::SlateDBError; -use crate::iter::{EmptyIterator, RowEntryIterator}; use crate::manifest::ManifestCore; -use crate::manifest::SsTableView; use crate::mem_table::WritableKVTable; -use crate::sst_iter::{SstIterator, SstIteratorOptions}; use crate::tablestore::TableStore; -use crate::utils::panic_string; -use crate::wal::{WalError, WalIterator as WalIteratorTrait, WalRows}; -use crate::RowEntry; -use async_trait::async_trait; -use log::error; -use std::collections::VecDeque; +#[cfg(test)] +use crate::wal::slatedb::iterator::{SlateDbWalIterator, SlateDbWalIteratorOptions}; +use crate::wal::WalIterator as WalIteratorTrait; +#[cfg(test)] use std::ops::Range; use std::sync::Arc; -use tokio::task; -use tokio::task::JoinHandle; - -pub(crate) struct WalIteratorOptions { - /// The number of SSTs to preload while replaying - pub(crate) sst_batch_size: usize, - - /// Options to pass through to underlying SST iterators - pub(crate) sst_iter_options: SstIteratorOptions, -} - -impl Default for WalIteratorOptions { - fn default() -> Self { - Self { - sst_batch_size: 4, - sst_iter_options: SstIteratorOptions::default(), - } - } -} pub(crate) struct WalReplayOptions { /// The target maximum number of bytes in each returned table. WAL replay only @@ -81,12 +56,12 @@ impl WalReplayIterator { pub(crate) fn range( wal_id_range: Range, db_state: &ManifestCore, - iterator_options: WalIteratorOptions, + iterator_options: SlateDbWalIteratorOptions, replay_options: WalReplayOptions, table_store: Arc, ) -> Result { let wal_iter = - WalIterator::range(wal_id_range, iterator_options, Arc::clone(&table_store))?; + SlateDbWalIterator::range(wal_id_range, iterator_options, Arc::clone(&table_store))?; Self::for_wal_iterator(Box::new(wal_iter), db_state, replay_options, table_store) } @@ -203,312 +178,9 @@ impl WalReplayIterator { } } -struct WalRowsCollector { - wal_id: u64, - iter: Box, - rows: Vec, - drained: bool, -} - -impl WalRowsCollector { - fn new(wal_id: u64, iter: Box) -> Self { - Self { - wal_id, - iter, - rows: vec![], - drained: false, - } - } - - async fn collect(&mut self) -> Result<(), WalError> { - loop { - match self.iter.next().await { - Ok(Some(row)) => self.rows.push(row), - Ok(None) => { - self.drained = true; - break Ok(()); - } - Err(err) if err.has_object_store_not_found() => { - break Err(WalError::WalTruncated(self.wal_id)); - } - Err(err) => { - break Err(err.into()); - } - } - } - } -} - -impl From for WalRows { - fn from(reader: WalRowsCollector) -> Self { - assert!(reader.drained); - WalRows { - last_consumed_wal_file_id: reader.wal_id, - rows: reader.rows, - } - } -} - -struct CurrentWalFile { - initialized: bool, - collector: Option, -} - -impl CurrentWalFile { - fn initial() -> Self { - Self { - initialized: false, - collector: None, - } - } - - fn initialized(&self) -> bool { - self.initialized - } - - async fn collect(&mut self) -> Result, WalError> { - assert!(self.initialized); - let Some(collector) = &mut self.collector else { - return Ok(None); - }; - collector.collect().await?; - let collector = self.collector.take().expect("unreachable"); - self.initialized = false; - Ok(Some(collector.into())) - } - - fn advance(&mut self, collector: WalRowsCollector) { - assert!(!self.initialized); - self.initialized = true; - self.collector = Some(collector); - } - - fn finish(&mut self) { - self.initialized = true; - self.collector = None; - } -} - -/// Iterates over the writes in a range of WAL files, preloading up to -/// `sst_batch_size` WAL SSTs concurrently. Returns the rows of one WAL file per -/// [`WalRows`], and verifies that files carry strictly increasing seq -/// ranges — the ordering callers rely on to split and tag memtables safely. -/// -/// Preloading only opens each WAL SST (footer, index, and any eagerly fetched -/// blocks); a file's rows are read out only when it is returned from -/// [`Self::next`], so at most one file's rows are materialized at a time. -pub(crate) struct WalIterator { - options: WalIteratorOptions, - /// Range of WAL IDs to iterate over - wal_id_range: Range, - table_store: Arc, - next_files: VecDeque>>, - next_wal_id: u64, - /// The greatest seq returned so far, used to verify that WAL files arrive - /// with strictly increasing seq ranges. - last_seq: Option, - /// Set once iteration has ended, either because the range was exhausted or - /// because an error was returned. - terminal_result: Option, WalError>>, - current_file: CurrentWalFile, -} - -impl WalIterator { - pub(crate) fn range( - wal_id_range: Range, - options: WalIteratorOptions, - table_store: Arc, - ) -> Result { - if options.sst_batch_size < 1 { - return Err(SlateDBError::InvalidSSTBatchSize(options.sst_batch_size)); - } - - let next_wal_id = wal_id_range.start; - Ok(WalIterator { - options, - wal_id_range, - table_store, - next_files: VecDeque::new(), - next_wal_id, - last_seq: None, - terminal_result: None, - current_file: CurrentWalFile::initial(), - }) - } - - fn maybe_spawn_open(&mut self) -> bool { - if !self.wal_id_range.contains(&self.next_wal_id) - || self.next_files.len() >= self.options.sst_batch_size - { - return false; - } - - let next_wal_id = self.next_wal_id; - self.next_wal_id += 1; - - async fn try_open_file_iter( - wal_id: u64, - sst_iter_options: SstIteratorOptions, - table_store: Arc, - ) -> Result { - let sst = match table_store.open_sst(&SsTableId::Wal(wal_id)).await { - Ok(sst) => sst, - Err(SlateDBError::EmptySSTable) => { - // Zero-byte WAL files are fence markers; replay them as empty WALs - // so the last replayed WAL ID still advances past the marker. - return Ok(WalRowsCollector::new( - wal_id, - Box::new(EmptyIterator::new()), - )); - } - Err(err) => return Err(err), - }; - let iter = SstIterator::new_owned_initialized( - .., - SsTableView::identity(sst), - Arc::clone(&table_store), - sst_iter_options, - ) - .await?; - // An unbounded, unfiltered scan over a WAL SST always yields an - // iterator. `None` means the file cannot be read, and replay must - // fail rather than silently end early and drop the remaining WALs. - let Some(iter) = iter else { - error!( - "could not construct row iterator over WAL SST. [wal_id={}]", - wal_id - ); - return Err(SlateDBError::InvalidDBState); - }; - Ok(WalRowsCollector::new(wal_id, Box::new(iter))) - } - - async fn open_file_iter( - wal_id: u64, - sst_iter_options: SstIteratorOptions, - table_store: Arc, - ) -> Result { - match try_open_file_iter(wal_id, sst_iter_options, table_store).await { - Ok(iter) => Ok(iter), - Err(err) if err.has_object_store_not_found() => Err(WalError::WalTruncated(wal_id)), - Err(err) => Err(err.into()), - } - } - - let handle = task::spawn(open_file_iter( - next_wal_id, - self.options.sst_iter_options.clone(), - Arc::clone(&self.table_store), - )); - self.next_files.push_back(handle); - true - } - - /// Await the next preloaded WAL file and return an iterator over its rows. - /// Returns `None` when there are no more files to read. - async fn load_next_file(&mut self) -> Result<(), WalError> { - if self.current_file.initialized() { - return Ok(()); - } - // await a mutable ref to the task so that next remains cancel-safe - // see https://docs.rs/tokio/latest/tokio/task/struct.JoinHandle.html#cancel-safety - let Some(join_handle) = self.next_files.front_mut() else { - self.current_file.finish(); - return Ok(()); - }; - let result = join_handle.await; - self.next_files.pop_front(); - match result { - Ok(result) => { - self.current_file.advance(result?); - Ok(()) - } - Err(join_err) => { - let task_name = format!("wal_replay[{:?}]", self.wal_id_range); - let msg = if let Ok(panic_err) = join_err.try_into_panic() { - format!( - "wal_replay task panicked unexpectedly. [task_name={}, panic={}]", - task_name, - panic_string(&panic_err), - ) - } else { - format!("wal_replay task cancelled. [task_name={}]", task_name) - }; - error!("{}", msg); - let error = Arc::from(Box::::from(msg)); - Err(WalError::InternalError(error)) - } - } - } - - fn terminate( - &mut self, - result: Result, WalError>, - ) -> Result, WalError> { - self.terminal_result = Some(result.clone()); - for task in self.next_files.drain(..) { - task.abort(); - } - result - } -} - -#[async_trait] -impl WalIteratorTrait for WalIterator { - /// Get the next set of writes from the WAL files in the range. Each returned - /// [`WalRows`] holds the rows of one WAL file; a WAL file with no rows - /// yields a batch with empty `rows`. Returns `None` once all WAL files in the - /// range have been read. It is an error if a WAL file in the range is not - /// present. Errors are returned only on calls that return no batch, so rows - /// read from earlier WAL files are never dropped with a later file's error. - async fn next(&mut self) -> Result, WalError> { - if let Some(result) = self.terminal_result.clone() { - return result; - } - - while self.maybe_spawn_open() {} - if let Err(err) = self.load_next_file().await { - return self.terminate(Err(err)); - } - match self.current_file.collect().await { - Err(err) => self.terminate(Err(err)), - Ok(None) => self.terminate(Ok(None)), - Ok(Some(rows)) => { - // Verify that WAL files carry strictly increasing seq ranges. Replay - // relies on this ordering to split and tag memtables safely: a commit seq - // spanning two WAL files, or files with overlapping seq ranges, would - // break recovery's (wal_id, seq) watermark filtering. - if let Some(min_seq) = rows.rows.iter().map(|row| row.seq).min() { - if let Some(last_seq) = self.last_seq { - if min_seq <= last_seq { - let msg = format!( - "WAL replay saw out-of-order seqs across WAL files. \ - [wal_id={}, min_seq={}, last_seq={}]", - rows.last_consumed_wal_file_id, min_seq, last_seq, - ); - error!("{}", &msg); - let error = - Arc::from(Box::::from(msg)); - return self.terminate(Err(WalError::InternalError(error))); - } - } - let max_seq = rows - .rows - .iter() - .map(|row| row.seq) - .max() - .expect("non-empty rows have a max seq"); - self.last_seq = Some(max_seq); - } - Ok(Some(rows)) - } - } - } -} - #[cfg(test)] mod tests { - use super::{WalIterator, WalIteratorOptions, WalReplayIterator, WalReplayOptions}; + use super::{SlateDbWalIteratorOptions, WalReplayIterator, WalReplayOptions}; use crate::block_cache_policy::BlockCachePolicy; use crate::bytes_range::BytesRange; use crate::db_state::SsTableId; @@ -531,7 +203,7 @@ mod tests { use rand::Rng; use std::cmp::min; use std::collections::btree_map::Iter; - use std::collections::{BTreeMap, BTreeSet, VecDeque}; + use std::collections::{BTreeMap, VecDeque}; use std::sync::Arc; struct ScriptedWalIterator { @@ -559,7 +231,7 @@ mod tests { Self::range( wal_id_range, db_state, - WalIteratorOptions::default(), + SlateDbWalIteratorOptions::default(), options, table_store, ) @@ -638,40 +310,6 @@ mod tests { assert!(replay_iter.next().await.unwrap().is_none()); } - #[tokio::test] - async fn should_repeat_terminal_error_for_wal_iterator() { - let table_store = test_table_store(); - let mut wal_iter = WalIterator::range( - 1..2, - WalIteratorOptions::default(), - Arc::clone(&table_store), - ) - .unwrap(); - - assert!(matches!( - wal_iter.next().await, - Err(WalError::WalTruncated(1)) - )); - assert!(matches!( - wal_iter.next().await, - Err(WalError::WalTruncated(1)) - )); - } - - #[tokio::test] - async fn should_repeat_terminal_none_for_wal_iterator() { - let table_store = test_table_store(); - let mut wal_iter = WalIterator::range( - 1..1, - WalIteratorOptions::default(), - Arc::clone(&table_store), - ) - .unwrap(); - - assert!(wal_iter.next().await.unwrap().is_none()); - assert!(wal_iter.next().await.unwrap().is_none()); - } - #[tokio::test] async fn should_use_last_consumed_wal_file_id_as_replay_watermark() { let table_store = test_table_store(); @@ -1062,115 +700,6 @@ mod tests { } } - #[tokio::test] - async fn should_return_atomic_wal_rows_in_increasing_seq_order() { - let table_store = test_table_store(); - // Each file contains out-of-order rows and a sequence that appears twice. - // The iterator must keep both rows for a sequence in one batch, while the - // sequence range of the second batch must follow the first. - let wal_entries = [ - vec![ - RowEntry::new_value(b"key_001", &[b'x'; 128], 2), - RowEntry::new_value(b"key_002", &[b'x'; 128], 1), - RowEntry::new_value(b"key_003", &[b'x'; 128], 2), - ], - vec![ - RowEntry::new_value(b"key_004", &[b'x'; 128], 4), - RowEntry::new_value(b"key_005", &[b'x'; 128], 3), - RowEntry::new_value(b"key_006", &[b'x'; 128], 4), - ], - ]; - let mut expected_rows = BTreeMap::new(); - let mut expected_rows_by_seq = BTreeMap::>::new(); - let wal_file_count = wal_entries.len() as u64; - for (file_index, entries) in wal_entries.iter().enumerate() { - let wal_id = file_index as u64 + 1; - for row in entries { - expected_rows.insert(row.key.clone(), (row.clone(), wal_id)); - expected_rows_by_seq - .entry(row.seq) - .or_default() - .insert(row.key.clone()); - } - } - for (index, entries) in wal_entries.into_iter().enumerate() { - let mut builder = table_store.wal_table_builder(); - for entry in entries { - builder.add(entry).await.unwrap(); - } - let encoded_sst = builder.build().await.unwrap(); - table_store - .write_sst(&SsTableId::Wal(index as u64 + 1), &encoded_sst) - .await - .unwrap(); - } - let mut wal_iter = WalIterator::range( - 1..(wal_file_count + 1), - WalIteratorOptions::default(), - Arc::clone(&table_store), - ) - .unwrap(); - - let mut returned_rows = BTreeMap::new(); - let mut previous_max_seq = None; - let mut last_consumed_wal_file_id = 0; - while let Some(batch) = wal_iter.next().await.unwrap() { - let batch_min_seq = batch.rows.iter().map(|r| r.seq).min().unwrap(); - let batch_max_seq = batch.rows.iter().map(|r| r.seq).max().unwrap(); - if let Some(previous_max_seq) = previous_max_seq { - assert!( - batch_min_seq > previous_max_seq, - "consecutive WAL batches have overlapping sequence ranges" - ); - } - previous_max_seq = Some(batch_max_seq); - - let mut batch_rows_by_seq = BTreeMap::>::new(); - for row in &batch.rows { - assert!( - returned_rows.insert(row.key.clone(), row.clone()).is_none(), - "row was returned more than once: {:?}", - row.key - ); - batch_rows_by_seq - .entry(row.seq) - .or_default() - .insert(row.key.clone()); - } - for (seq, batch_rows) in batch_rows_by_seq { - assert_eq!( - expected_rows_by_seq.get(&seq), - Some(&batch_rows), - "rows for seq {seq} were split across WAL batches" - ); - } - - assert!( - batch.last_consumed_wal_file_id >= last_consumed_wal_file_id, - "consumed WAL file watermark moved backwards" - ); - assert!(batch.last_consumed_wal_file_id <= wal_file_count); - for wal_id in 1..=batch.last_consumed_wal_file_id { - let file_fully_returned = - expected_rows.iter().all(|(key, (_, expected_wal_id))| { - *expected_wal_id != wal_id || returned_rows.contains_key(key) - }); - assert!( - file_fully_returned, - "WAL file {wal_id} was marked consumed before all its rows were returned" - ); - } - last_consumed_wal_file_id = batch.last_consumed_wal_file_id; - } - - let expected_returned_rows = expected_rows - .into_iter() - .map(|(key, (row, _wal_id))| (key, row)) - .collect(); - assert_eq!(returned_rows, expected_returned_rows); - assert_eq!(last_consumed_wal_file_id, wal_file_count); - } - #[tokio::test] async fn should_only_replay_wals_after_last_l0_flushed_wal_id() { let table_store = test_table_store(); From 9fa88d6aaf2f02ebf8146580c316f51c1bfca63b Mon Sep 17 00:00:00 2001 From: Tyler Rockwood Date: Mon, 17 Aug 2026 13:15:25 -0500 Subject: [PATCH 36/65] union: validate non-overlapping key ranges per segment (#2030) --- examples/src/rescaling.rs | 64 +---- rfcs/0004-checkpoints.md | 2 + rfcs/0024-segment-oriented-compaction.md | 5 +- slatedb/src/admin.rs | 2 + slatedb/src/clone.rs | 106 ++++++++ slatedb/src/manifest/mod.rs | 295 +++++++++++++++++++---- slatedb/src/test_utils.rs | 29 +++ 7 files changed, 395 insertions(+), 108 deletions(-) diff --git a/examples/src/rescaling.rs b/examples/src/rescaling.rs index 10a8fc2c73..663968196d 100644 --- a/examples/src/rescaling.rs +++ b/examples/src/rescaling.rs @@ -9,6 +9,10 @@ //! [`CheckpointScope::Durable`] before cloning. The same APIs work for //! segmented and non-segmented stores. //! +//! Union requires its sources to be non-overlapping. For a segmented store +//! that rule applies per segment, so the two shards below merge in one call +//! even though each holds part of both segments. +//! //! Clone construction uses: //! [`AdminBuilder::new`] → [`Admin::create_clone_builder_from_source`] → //! [`CloneBuilder::with_source`] → [`CloneBuilder::build`]. @@ -95,37 +99,6 @@ fn right_tenant_in_segment(prefix: &[u8]) -> ProjectionRange { ) } -/// Contiguous `data/…` / `idx/…` slices for union. A tenant shard that holds -/// both segments has a bounding range spanning `data/…` through `idx/…`, so -/// left and right overlap unless re-sliced before merge. -fn data_left_range() -> ProjectionRange { - ( - Bound::Unbounded, - Bound::Excluded(Bytes::from_static(b"idx")), - ) -} - -fn data_right_range() -> ProjectionRange { - ( - Bound::Included(kind_tenant(b"data", SPLIT_TENANT)), - Bound::Excluded(Bytes::from_static(b"idx")), - ) -} - -fn idx_left_range() -> ProjectionRange { - ( - Bound::Included(Bytes::from_static(b"idx")), - Bound::Excluded(kind_tenant(b"idx", SPLIT_TENANT)), - ) -} - -fn idx_right_range() -> ProjectionRange { - ( - Bound::Included(kind_tenant(b"idx", SPLIT_TENANT)), - Bound::Unbounded, - ) -} - #[tokio::main] async fn main() -> anyhow::Result<()> { let object_store = Arc::new(InMemory::new()); @@ -215,8 +188,6 @@ async fn rescale_segmented(object_store: Arc) -> anyhow::Result<()> { let root_path = "/tmp/slatedb_rescaling/segmented/root"; let left_path = "/tmp/slatedb_rescaling/segmented/left"; let right_path = "/tmp/slatedb_rescaling/segmented/right"; - let merged_data_path = "/tmp/slatedb_rescaling/segmented/merged_data"; - let merged_idx_path = "/tmp/slatedb_rescaling/segmented/merged_idx"; let merged_path = "/tmp/slatedb_rescaling/segmented/merged"; let db = Db::builder(root_path, object_store.clone()) @@ -289,31 +260,14 @@ async fn rescale_segmented(object_store: Arc) -> anyhow::Result<()> { left.close().await?; right.close().await?; - // Scale down: union needs non-overlapping bounding ranges, so re-slice each - // shard into contiguous `data/…` and `idx/…` halves, then union those. - create_clone( - merged_data_path, - vec![ - CloneSourceSpec::new(left_path).with_projection_range(data_left_range()), - CloneSourceSpec::new(right_path).with_projection_range(data_right_range()), - ], - object_store.clone(), - ) - .await?; - create_clone( - merged_idx_path, - vec![ - CloneSourceSpec::new(left_path).with_projection_range(idx_left_range()), - CloneSourceSpec::new(right_path).with_projection_range(idx_right_range()), - ], - object_store.clone(), - ) - .await?; + // Scale down: one union merges both shards. Union requires the sources to + // be non-overlapping per segment, not overall — each shard holds the lower + // or upper zoos of both `data` and `idx`, so no segment is claimed twice. create_clone( merged_path, vec![ - CloneSourceSpec::new(merged_data_path), - CloneSourceSpec::new(merged_idx_path), + CloneSourceSpec::new(left_path), + CloneSourceSpec::new(right_path), ], object_store.clone(), ) diff --git a/rfcs/0004-checkpoints.md b/rfcs/0004-checkpoints.md index 69868e0977..d6fd4eac71 100644 --- a/rfcs/0004-checkpoints.md +++ b/rfcs/0004-checkpoints.md @@ -748,6 +748,8 @@ The union process works as follows: of all its L0 and compacted SSTs and optionally intersected with `visible_range`s for each manifest if they are provided by the user. If any two manifests have intersecting key ranges, the operation fails. If the `visible_ranges` are explicitly provided then validate that they are adjacent. + Segmented manifests apply this check per segment rather than to the manifest as a whole; see + [RFC 0024](./0024-segment-oriented-compaction.md#interaction-with-projection-and-union). 3. Merge the contents of all input manifests: - `external_dbs` entries from all input manifests are merged and deduplicated by `(path, source_checkpoint_id)`. `external_dbs` with the same `(path, source_checkpoint_id)` originated from the diff --git a/rfcs/0024-segment-oriented-compaction.md b/rfcs/0024-segment-oriented-compaction.md index 67fd2922be..4fe6afdb0e 100644 --- a/rfcs/0024-segment-oriented-compaction.md +++ b/rfcs/0024-segment-oriented-compaction.md @@ -336,10 +336,11 @@ The extractor must be configured when the database is first created, or never co **Projection.** For each segment in `segments`, apply the same view-intersection rules as for the unsegmented `l0` and `compacted` lists: drop SST views whose effective range lies fully outside the projection range, and tag boundary views with a `visible_range`. Segments whose views are all excluded are removed from `segments`. The `segment_extractor_name` field is preserved unchanged. After projection a segment's effective range may be narrower than `[prefix, prefix++)`; this is benign, as `visible_range` enforcement on each view governs read and write access. -**Union.** Union of N segmented manifests adds two preconditions on top of those in [RFC 0004](./0004-checkpoints.md#union): +**Union.** Union of N segmented manifests relaxes one precondition from [RFC 0004](./0004-checkpoints.md#union) and adds two: +- The non-overlapping key range precondition applies per segment, not to each source's manifest as a whole. A read routes to exactly one segment, and each segment's chain is built only from that prefix's entries, so sources that overlap across *different* segments never collide; only sources contributing to the same prefix must be disjoint. This is what makes a segmented database rescalable along a dimension the segment prefix does not lead with. Shards holding `data/{tenant}` and `idx/{tenant}` rows, split by tenant, each span both segments — their bounding ranges overlap while no single segment does, so they union in one operation instead of through staged re-slicing clones. - All sources must share the same `segment_extractor_name` exactly — every source `None`, or every source the same `Some(name)`. Mixed configurations are rejected. Although unioned ranges are disjoint, we don't know that unsegmented data from a no-extractor source will remain unsegmented after union: a key persisted in `core.tree` may match an extractor prefix carried over from another source, and a future read of that key would route through the extractor to a segment that does not contain it, dropping the value. A future extension can relax this once a per-key check confirms unsegmented data does not match any extractor prefix; for now we require exact agreement. -- The combined set of segment prefixes across all sources must form an antichain (no prefix is a proper prefix of another). This usually follows from the existing non-overlapping key range precondition, but is checked explicitly to defend against stale extractor-name matches. +- The combined set of segment prefixes across all sources must form an antichain (no prefix is a proper prefix of another). With the key range precondition now scoped per segment, this no longer follows from it at all, and it also defends against stale extractor-name matches. For each segment prefix in the inputs: diff --git a/slatedb/src/admin.rs b/slatedb/src/admin.rs index 49ea0182bb..81ba412d71 100644 --- a/slatedb/src/admin.rs +++ b/slatedb/src/admin.rs @@ -761,6 +761,8 @@ impl Admin { /// [`CloneBuilder::with_source`]: each source must carry its own per-source range so that /// [`crate::manifest::Manifest::cloned_from_union`] sees non-overlapping effective ranges. /// + /// Segmented sources only have to be non-overlapping within each segment that they share. + /// /// # Examples /// /// ``` diff --git a/slatedb/src/clone.rs b/slatedb/src/clone.rs index 62116f7340..3d9127c8bc 100644 --- a/slatedb/src/clone.rs +++ b/slatedb/src/clone.rs @@ -2137,6 +2137,112 @@ mod tests { clone_db.close().await.unwrap(); } + #[cfg(feature = "wal_disable")] + #[tokio::test] + async fn should_union_segmented_shards_that_each_span_every_segment() { + // Rescale-down of a store keyed `data/{tenant}/…` and + // `idx/{tenant}/…`, sharded by tenant. Each shard holds part of both + // segments, so the shards' overall key ranges overlap — + // `data/metro…` sorts below `idx/bronx…` — while neither segment + // does. One union call must merge them; no per-source projection and + // no staged re-slicing clones are needed. + let object_store: Arc = Arc::new(InMemory::new()); + let parent_path_a = Path::from("/tmp/test_parent_seg_interleaved_a"); + let parent_path_b = Path::from("/tmp/test_parent_seg_interleaved_b"); + let clone_path = Path::from("/tmp/test_clone_seg_interleaved"); + let extractor = Arc::new(test_utils::DataIdxPrefixExtractor); + let settings = wal_disabled_settings(); + + fn shard(tenants: [&str; 2]) -> BTreeMap { + let mut table = BTreeMap::new(); + for tenant in tenants { + table.insert( + Bytes::from(format!("data/{}/animal/lion-1", tenant)), + Bytes::from(format!("{} lion", tenant)), + ); + table.insert( + Bytes::from(format!("idx/{}/owner/alice/lion-1", tenant)), + Bytes::new(), + ); + } + table + } + let table_a = shard(["bronx", "lincoln"]); + let table_b = shard(["metro", "oakland"]); + + build_segmented_parent( + &parent_path_a, + object_store.clone(), + extractor.clone(), + settings.clone(), + &table_a, + ) + .await; + build_segmented_parent( + &parent_path_b, + object_store.clone(), + extractor.clone(), + settings.clone(), + &table_b, + ) + .await; + + run_segmented_clone( + vec![ + CloneSourceSpec::new(parent_path_a.clone()), + CloneSourceSpec::new(parent_path_b.clone()), + ], + &clone_path, + object_store.clone(), + None, + ) + .await; + + // Both shards contribute an L0 SST to each of the two segments. + let store = ManifestStore::new(&clone_path, object_store.clone()); + let stored = store.read_latest_manifest().await.unwrap(); + assert_eq!( + stored.manifest.core.segment_extractor_name.as_deref(), + Some("data-idx") + ); + let segments: Vec<(Bytes, usize)> = stored + .manifest + .core + .segments + .iter() + .map(|s| (s.prefix.clone(), s.tree.l0.len())) + .collect(); + assert_eq!( + segments, + vec![ + (Bytes::from_static(b"data"), 2), + (Bytes::from_static(b"idx"), 2) + ] + ); + assert_eq!(stored.manifest.external_dbs.len(), 2); + + let mut expected: BTreeMap = table_a.clone(); + expected.extend(table_b.clone()); + + let clone_db = + open_segmented_clone(&clone_path, object_store.clone(), extractor, settings).await; + let mut full_iter = clone_db.scan(..).await.unwrap(); + test_utils::assert_ranged_db_scan(&expected, .., IterationOrder::Ascending, &mut full_iter) + .await; + // Each segment routes reads across both shards' contributions. + assert_segment_prefix_scan(&clone_db, &expected, b"data", b"datb").await; + assert_segment_prefix_scan(&clone_db, &expected, b"idx", b"idy").await; + for (key, value) in &expected { + assert_eq!( + clone_db.get(key).await.unwrap().as_ref(), + Some(value), + "key={:?}", + key + ); + } + clone_db.close().await.unwrap(); + } + #[cfg(feature = "wal_disable")] #[tokio::test] async fn should_union_projected_segmented_dbs() { diff --git a/slatedb/src/manifest/mod.rs b/slatedb/src/manifest/mod.rs index f8af20347e..2558f39f13 100644 --- a/slatedb/src/manifest/mod.rs +++ b/slatedb/src/manifest/mod.rs @@ -22,6 +22,10 @@ pub(crate) mod store; pub use crate::db_state::{SortedRun, SsTableHandle, SsTableId, SsTableInfo, SsTableView}; +/// The per-source trees a union merges into each segment, keyed by segment +/// prefix and ordered within a segment by that segment's start key. +type SegmentContributors<'a> = BTreeMap>; + /// Per-LSM-tree state. Shared shape between the unsegmented tree (held directly /// on `ManifestCore`) and each named segment held in `ManifestCore::segments`. #[derive(Clone, Default, PartialEq, Serialize, Debug)] @@ -66,6 +70,14 @@ impl LsmTreeState { self.l0.is_empty() && self.compacted.is_empty() } + /// Iterate every SST view referenced by this tree — L0 views followed by + /// the views of every sorted run. + pub(crate) fn sst_views(&self) -> impl Iterator { + self.l0 + .iter() + .chain(self.compacted.iter().flat_map(|sr| sr.sst_views().iter())) + } + /// Total number of SST views referenced by this tree — L0 plus every /// SST in every sorted run. Used by the read path to size scan /// parallelism. @@ -551,11 +563,7 @@ impl ManifestCore { /// Iterate every SST view referenced by this manifest — L0 views and /// sorted-run views across the unsegmented tree and every segment. pub(crate) fn all_sst_views(&self) -> impl Iterator { - self.trees().flat_map(|tree| { - tree.l0 - .iter() - .chain(tree.compacted.iter().flat_map(|sr| sr.sst_views().iter())) - }) + self.trees().flat_map(|tree| tree.sst_views()) } /// Compare a configured WAL-store URI against the manifest's @@ -1290,38 +1298,90 @@ impl Manifest { Ok(()) } - /// Extractor-configured case. Build a per-prefix accumulator from - /// every source's `core.segments`, validate the antichain, and write - /// the result into `core.segments`. Empty entries (no L0, no compacted - /// runs) are dropped — the unioned manifest should not carry - /// placeholders. Watermarks are intentionally not carried over: the - /// unioned manifest is a fresh DB that begins compaction tracking from - /// scratch. - fn build_segmented_lsm_state( - core: &mut ManifestCore, - sources: &[&CloneSource], - ) -> Result<(), SlateDBError> { - let mut segments_by_prefix: BTreeMap = BTreeMap::new(); + /// Extractor-configured case. Concatenate the per-source trees that + /// [`Self::group_segments_for_union`] collected for each prefix into one + /// tree per segment. Segments holding no SSTs are already absent from + /// `segments`, so the unioned manifest carries no placeholders. + /// Watermarks are intentionally not carried over: the unioned manifest is + /// a fresh DB that begins compaction tracking from scratch. + fn build_segmented_lsm_state(core: &mut ManifestCore, segments: SegmentContributors) { + core.segments = segments + .into_iter() + .map(|(prefix, trees)| { + let mut merged = LsmTreeState::default(); + for tree in trees { + merged.l0.extend(tree.l0.iter().cloned()); + merged.compacted.extend(tree.compacted.iter().cloned()); + } + Segment { + prefix, + tree: Arc::new(merged), + } + }) + .collect(); + } + + /// Collect the trees that each source contributes to each segment. Within + /// a segment the trees are in key order, so the merged segment's `l0` and + /// `compacted` lists come out sorted — the same thing the manifest-wide + /// sort does for the unsegmented tree. + /// + /// Two things are rejected. If one segment prefix is a prefix of another, + /// the two segments cover some of the same keys. If two sources hold the + /// same keys inside one segment, the merged segment cannot say which + /// source owns them. + /// + /// A segment with no SSTs contributes nothing, so it is left out. + fn group_segments_for_union<'a>( + sources: &[&'a CloneSource], + ) -> Result, SlateDBError> { + let all_prefixes: BTreeSet<&Bytes> = sources + .iter() + .flat_map(|source| source.manifest.core.segments.iter().map(|seg| &seg.prefix)) + .collect(); + Self::ensure_union_prefix_antichain(all_prefixes)?; + + let mut by_prefix: BTreeMap> = BTreeMap::new(); for source in sources { for segment in &source.manifest.core.segments { - let entry = segments_by_prefix - .entry(segment.prefix.clone()) - .or_default(); - entry.l0.extend(segment.tree.l0.iter().cloned()); - entry - .compacted - .extend(segment.tree.compacted.iter().cloned()); + if let Some(range) = Self::bounding_range(segment.tree.sst_views()) { + by_prefix + .entry(segment.prefix.clone()) + .or_default() + .push((range, segment.tree.as_ref())); + } + } + } + + let mut grouped = SegmentContributors::new(); + for (prefix, mut members) in by_prefix { + members.sort_by_key(|(range, _)| range.comparable_start_bound().cloned()); + let ranges: Vec = members.iter().map(|(range, _)| range.clone()).collect(); + Self::ensure_disjoint_ranges(Some(&prefix), &ranges)?; + grouped.insert(prefix, members.into_iter().map(|(_, tree)| tree).collect()); + } + Ok(grouped) + } + + /// Reject overlapping source ranges. `ranges` must be sorted by start + /// bound. `segment` names the segment under check and is omitted from the + /// message for the unsegmented tree. + fn ensure_disjoint_ranges( + segment: Option<&Bytes>, + ranges: &[BytesRange], + ) -> Result<(), SlateDBError> { + for pair in ranges.windows(2) { + if pair[1].intersect(&pair[0]).is_some() { + let scope = match segment { + Some(prefix) => format!(" in segment `{:?}`", prefix), + None => String::new(), + }; + return Err(SlateDBError::InvalidUnion(format!( + "clone sources have overlapping key ranges{}. ranges=`{:?}`", + scope, ranges + ))); } } - Self::ensure_union_prefix_antichain(segments_by_prefix.keys())?; - core.segments = segments_by_prefix - .into_iter() - .filter(|(_, tree)| !tree.is_empty()) - .map(|(prefix, tree)| Segment { - prefix, - tree: Arc::new(tree), - }) - .collect(); Ok(()) } @@ -1401,30 +1461,17 @@ impl Manifest { } ranges.sort_by_key(|(_, range)| range.comparable_start_bound().cloned()); - // Ensure source key ranges are non-overlapping. Surfaces as a typed - // error since the source set is user-supplied. - let mut previous_range = None; - for (_, range) in ranges.iter() { - if let Some(previous_range) = previous_range { - if range.intersect(previous_range).is_some() { - let all: Vec = ranges.iter().map(|(_, r)| (*r).clone()).collect(); - return Err(SlateDBError::InvalidUnion(format!( - "clone sources have overlapping key ranges. ranges=`{:?}`", - all - ))); - } - } - previous_range = Some(range); - } - let ordered_sources: Vec<&CloneSource> = ranges.iter().map(|(s, _)| *s).collect(); let mut core = ManifestCore::new(); core.segment_extractor_name = Self::ensure_consistent_segment_extractor(&sources)?; if core.segment_extractor_name.is_none() { + let all: Vec = ranges.iter().map(|(_, r)| (*r).clone()).collect(); + Self::ensure_disjoint_ranges(None, &all)?; Self::build_unsegmented_lsm_state(&mut core, &ordered_sources)?; } else { - Self::build_segmented_lsm_state(&mut core, &ordered_sources)?; + let segments = Self::group_segments_for_union(&ordered_sources)?; + Self::build_segmented_lsm_state(&mut core, segments); } Self::renumber_union_sorted_runs(&mut core); @@ -1485,9 +1532,15 @@ impl Manifest { } fn range(&self) -> Option { + Self::bounding_range(self.core.all_sst_views()) + } + + /// Smallest range covering every view in `views`, or `None` when `views` + /// is empty. + fn bounding_range<'a>(views: impl Iterator) -> Option { let mut start_bound = None; let mut end_bound = None; - for sst in self.core.all_sst_views() { + for sst in views { let range = sst.compacted_effective_range(); start_bound = start_bound .map(|b| min(b, range.comparable_start_bound())) @@ -4306,6 +4359,146 @@ mod tests { assert_ne!(sr_ids[0], sr_ids[1]); } + /// Build a segmented manifest whose segments each hold a single L0 view + /// over the given range. `segments` is `(prefix, first_entry, range)` and + /// must be ordered by prefix, as `ManifestCore::segments` is. + fn segmented_shard_manifest( + extractor_name: &str, + segments: Vec<(&'static [u8], &'static str, BytesRange)>, + ) -> Manifest { + let mut core = ManifestCore::new(); + core.segment_extractor_name = Some(extractor_name.to_string()); + core.segments = segments + .into_iter() + .map(|(prefix, first_entry, range)| { + let sst = SsTableId::Compacted(Ulid::new()); + Segment { + prefix: Bytes::from_static(prefix), + tree: Arc::new(LsmTreeState { + last_compacted_l0_sst_view_id: None, + last_compacted_l0_sst_id: None, + l0: VecDeque::from(vec![SsTableView::new_projected( + sst.unwrap_compacted_id(), + SsTableHandle::new( + sst, + SST_FORMAT_VERSION_LATEST, + SsTableInfo { + first_entry: Some(Bytes::from_static(first_entry.as_bytes())), + ..SsTableInfo::default() + }, + ), + Some(range), + )]), + compacted: vec![], + }), + } + }) + .collect(); + Manifest::initial(core) + } + + fn union_of(manifests: Vec) -> Result { + Manifest::cloned_from_union( + manifests + .into_iter() + .enumerate() + .map(|(i, manifest)| CloneSource { + manifest, + path: Path::from(format!("/tmp/db{}", i)), + checkpoint: new_checkpoint(Uuid::new_v4()), + }) + .collect(), + Arc::new(DbRand::default()), + ) + } + + /// Effective range of each L0 view in the named segment, in list order. + fn segment_l0_ranges(manifest: &Manifest, prefix: &[u8]) -> Vec { + manifest + .core + .segments + .iter() + .find(|s| s.prefix == prefix) + .expect("segment present") + .tree + .l0 + .iter() + .map(|view| view.compacted_effective_range().clone()) + .collect() + } + + #[test] + fn test_union_validates_key_ranges_per_segment() { + // Two tenant shards of a store keyed `data/{tenant}` and + // `idx/{tenant}`. Each shard holds part of both segments, so the + // shards' bounding ranges overlap — `data/metro` sorts below + // `idx/bronx` — while neither segment does. A read routes to exactly + // one segment, so the union is unambiguous and must be accepted. + // + // The shards also lead in different segments: `left` holds the lower + // `data` keys, `right` the lower `idx` keys. Each segment's list must + // follow its own key order rather than the manifest-wide source + // order, so a source's entries stay contiguous and ascending within + // the segment, as they do in the unsegmented case. + fn shard(data: BytesRange, idx: BytesRange) -> Manifest { + segmented_shard_manifest("kind", vec![(b"data", "data", data), (b"idx", "idx", idx)]) + } + let left = shard( + BytesRange::from_ref("data/bronx".."data/metro"), + BytesRange::from_ref("idx/metro".."idx/zzz"), + ); + let right = shard( + BytesRange::from_ref("data/metro".."data/zzz"), + BytesRange::from_ref("idx/bronx".."idx/metro"), + ); + + // The precondition the old manifest-wide check enforced is genuinely + // violated here; only the per-segment check makes this union legal. + let left_range = left.range().expect("left range"); + assert!(left_range + .intersect(&right.range().expect("right range")) + .is_some()); + + let union = union_of(vec![left, right]).expect("union of disjoint segments"); + + assert_eq!(union.core.segment_extractor_name.as_deref(), Some("kind")); + assert_eq!( + segment_l0_ranges(&union, b"data"), + vec![ + BytesRange::from_ref("data/bronx".."data/metro"), + BytesRange::from_ref("data/metro".."data/zzz"), + ] + ); + assert_eq!( + segment_l0_ranges(&union, b"idx"), + vec![ + BytesRange::from_ref("idx/bronx".."idx/metro"), + BytesRange::from_ref("idx/metro".."idx/zzz"), + ] + ); + + // Overlap *within* a shared segment is still rejected: the merged + // segment's read chain could not say which source owns the key. + let overlapping = shard( + BytesRange::from_ref("data/lincoln".."data/zzz"), + BytesRange::from_ref("idx/bronx".."idx/metro"), + ); + let err = union_of(vec![ + shard( + BytesRange::from_ref("data/bronx".."data/metro"), + BytesRange::from_ref("idx/metro".."idx/zzz"), + ), + overlapping, + ]) + .expect_err("overlap within `data`"); + let SlateDBError::InvalidUnion(msg) = err else { + panic!("expected InvalidUnion, got {:?}", err); + }; + // The message names the offending segment, not just the ranges. + assert!(msg.contains("in segment"), "{}", msg); + assert!(msg.contains("data"), "{}", msg); + } + #[test] fn test_union_unsegmented_sources_land_in_core_tree() { // Two sources with no extractor configured. The unioned manifest diff --git a/slatedb/src/test_utils.rs b/slatedb/src/test_utils.rs index c6d3573b7c..6be520b303 100644 --- a/slatedb/src/test_utils.rs +++ b/slatedb/src/test_utils.rs @@ -1395,6 +1395,35 @@ impl crate::prefix_extractor::PrefixExtractor for FixedThreeBytePrefixExtractor } } +/// Test extractor that segments on the leading `data` / `idx` path +/// component, modelling a store that keeps bulky records in one segment and +/// a smaller index over them in another. Tenants live in the *next* +/// component (`data/{tenant}/…`), so a tenant's rows are split across both +/// segments rather than forming one contiguous key range. +// Only the union tests in `clone.rs` use this, and those need `wal_disable`. +#[cfg(feature = "wal_disable")] +#[derive(Debug)] +pub(crate) struct DataIdxPrefixExtractor; + +#[cfg(feature = "wal_disable")] +impl crate::prefix_extractor::PrefixExtractor for DataIdxPrefixExtractor { + fn name(&self) -> &str { + "data-idx" + } + fn prefix_len(&self, target: &crate::prefix_extractor::PrefixTarget) -> Option { + let key: &[u8] = match target { + crate::prefix_extractor::PrefixTarget::Point(b) + | crate::prefix_extractor::PrefixTarget::Prefix(b) => b.as_ref(), + }; + for kind in [b"data".as_slice(), b"idx".as_slice()] { + if key == kind || key.starts_with(&[kind, b"/".as_slice()].concat()) { + return Some(kind.len()); + } + } + None + } +} + /// Test extractor that deliberately violates the /// [`crate::prefix_extractor::PrefixExtractor`] `Point` invariant by /// returning different prefix lengths for keys that share a common From 2a2fdd146a95c0e2eb6cad84519b019f30a5ecfb Mon Sep 17 00:00:00 2001 From: Rohan Date: Wed, 19 Aug 2026 01:00:27 -0400 Subject: [PATCH 37/65] [rfc-30 8/N]: support live streaming from SlateDbWalIterator and use for CDC (#2032) --- bindings/go/uniffi/doc.go | 23 +- bindings/go/uniffi/slatedb.go | 982 +++++++++--------- bindings/go/uniffi/slatedb.h | 158 ++- bindings/go/uniffi/slatedb_test.go | 303 +++--- .../slatedb/uniffi/SlateDbWalReaderTest.java | 174 ++-- .../java/io/slatedb/uniffi/TestSupport.java | 29 +- bindings/node/tests/support.mjs | 36 +- bindings/node/tests/wal-reader.test.mjs | 138 ++- bindings/python/tests/conftest.py | 28 +- bindings/python/tests/test_wal_reader.py | 115 +- bindings/uniffi/src/error.rs | 57 + bindings/uniffi/src/lib.rs | 2 +- bindings/uniffi/src/wal_reader.rs | 226 ++-- examples/src/change_data_capture.rs | 78 +- rfcs/0030-pluggable-wal.md | 39 +- slatedb/src/db.rs | 24 +- slatedb/src/db_reader.rs | 79 +- slatedb/src/wal/mod.rs | 12 +- slatedb/src/wal/slatedb/iterator.rs | 274 ++++- slatedb/src/wal/slatedb/reader.rs | 713 ++++++++++++- slatedb/src/wal/slatedb/writer_init.rs | 9 +- slatedb/src/wal_replay.rs | 12 +- .../docs/docs/design/change-data-capture.mdx | 122 +-- 23 files changed, 2284 insertions(+), 1349 deletions(-) diff --git a/bindings/go/uniffi/doc.go b/bindings/go/uniffi/doc.go index a4672ff4de..cd8744dfaf 100644 --- a/bindings/go/uniffi/doc.go +++ b/bindings/go/uniffi/doc.go @@ -135,13 +135,15 @@ // interface. Rust-side logging can be forwarded into Go code with // [InitLogging] and a [LogCallback]. // -// # WAL Inspection +// # Change Data Capture // -// [NewWalReader] opens a [WalReader] for inspecting WAL files under a database -// path. [WalReader.List] enumerates [WalFile] handles, [WalFile.Metadata] -// returns object-store metadata, and [WalFile.Iterator] returns a -// [WalFileIterator] that yields raw [RowEntry] values. This is primarily useful -// for debugging, diagnostics, and low-level tooling. +// [NewSlateDbWalReader] opens a [SlateDbWalReader] for live WAL streaming. +// Call [SlateDbWalReader.Iterator] once with the first unconsumed WAL file ID, +// then keep calling [SlateDbWalIterator.Next]. The iterator waits and polls +// internally at the current tail. Persist every [WalRows.LastConsumedWalFileId], +// including empty fence batches, and resume from the following ID after a +// restart. [SlateDbWalReader.LastWalFileId] is available when a snapshot of the +// current tail is useful, but is not needed to drive the stream. // // # Errors // @@ -159,8 +161,8 @@ // // Most exported handle types own a Rust-side resource and provide an explicit // `Destroy` method, including [ObjectStore], [DbBuilder], [Db], [DbReader], -// [DbSnapshot], [DbTransaction], [DbIterator], [WalReader], [WalFile], -// [WalFileIterator], [Settings], and [WriteBatch]. +// [DbSnapshot], [DbTransaction], [DbIterator], [SlateDbWalReader], +// [SlateDbWalIterator], [Settings], and [WriteBatch]. // // These handles install Go finalizers, but callers should not rely on garbage // collection for timely cleanup. Prefer calling `Destroy` explicitly when a @@ -169,6 +171,7 @@ // binding handle. // // Builders are single-use after `Build`. [WriteBatch] is single-use after -// `Write`. Iterator `Next` methods return `nil` when exhausted, and transaction -// commit methods may return `nil` when no write was emitted. +// `Write`. Bounded iterator `Next` methods return `nil` when exhausted; the live +// [SlateDbWalIterator] instead waits at the current tail. Transaction commit +// methods may return `nil` when no write was emitted. package slatedb diff --git a/bindings/go/uniffi/slatedb.go b/bindings/go/uniffi/slatedb.go index a16849adcf..69b251b3c9 100644 --- a/bindings/go/uniffi/slatedb.go +++ b/bindings/go/uniffi/slatedb.go @@ -1598,74 +1598,29 @@ func uniffiCheckChecksums() { } { checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { - return C.uniffi_slatedb_uniffi_checksum_method_walfile_id() + return C.uniffi_slatedb_uniffi_checksum_method_slatedbwaliterator_next() }) - if checksum != 62512 { + if checksum != 46461 { // If this happens try cleaning and rebuilding your project - panic("slatedb: uniffi_slatedb_uniffi_checksum_method_walfile_id: UniFFI API checksum mismatch") + panic("slatedb: uniffi_slatedb_uniffi_checksum_method_slatedbwaliterator_next: UniFFI API checksum mismatch") } } { checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { - return C.uniffi_slatedb_uniffi_checksum_method_walfile_iterator() + return C.uniffi_slatedb_uniffi_checksum_method_slatedbwalreader_iterator() }) - if checksum != 46880 { + if checksum != 10327 { // If this happens try cleaning and rebuilding your project - panic("slatedb: uniffi_slatedb_uniffi_checksum_method_walfile_iterator: UniFFI API checksum mismatch") + panic("slatedb: uniffi_slatedb_uniffi_checksum_method_slatedbwalreader_iterator: UniFFI API checksum mismatch") } } { checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { - return C.uniffi_slatedb_uniffi_checksum_method_walfile_metadata() + return C.uniffi_slatedb_uniffi_checksum_method_slatedbwalreader_last_wal_file_id() }) - if checksum != 45103 { + if checksum != 40190 { // If this happens try cleaning and rebuilding your project - panic("slatedb: uniffi_slatedb_uniffi_checksum_method_walfile_metadata: UniFFI API checksum mismatch") - } - } - { - checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { - return C.uniffi_slatedb_uniffi_checksum_method_walfile_next_file() - }) - if checksum != 56800 { - // If this happens try cleaning and rebuilding your project - panic("slatedb: uniffi_slatedb_uniffi_checksum_method_walfile_next_file: UniFFI API checksum mismatch") - } - } - { - checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { - return C.uniffi_slatedb_uniffi_checksum_method_walfile_next_id() - }) - if checksum != 48353 { - // If this happens try cleaning and rebuilding your project - panic("slatedb: uniffi_slatedb_uniffi_checksum_method_walfile_next_id: UniFFI API checksum mismatch") - } - } - { - checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { - return C.uniffi_slatedb_uniffi_checksum_method_walfileiterator_next() - }) - if checksum != 51490 { - // If this happens try cleaning and rebuilding your project - panic("slatedb: uniffi_slatedb_uniffi_checksum_method_walfileiterator_next: UniFFI API checksum mismatch") - } - } - { - checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { - return C.uniffi_slatedb_uniffi_checksum_method_walreader_get() - }) - if checksum != 11510 { - // If this happens try cleaning and rebuilding your project - panic("slatedb: uniffi_slatedb_uniffi_checksum_method_walreader_get: UniFFI API checksum mismatch") - } - } - { - checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { - return C.uniffi_slatedb_uniffi_checksum_method_walreader_list() - }) - if checksum != 43661 { - // If this happens try cleaning and rebuilding your project - panic("slatedb: uniffi_slatedb_uniffi_checksum_method_walreader_list: UniFFI API checksum mismatch") + panic("slatedb: uniffi_slatedb_uniffi_checksum_method_slatedbwalreader_last_wal_file_id: UniFFI API checksum mismatch") } } { @@ -1895,11 +1850,38 @@ func uniffiCheckChecksums() { } { checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { - return C.uniffi_slatedb_uniffi_checksum_constructor_walreader_new() + return C.uniffi_slatedb_uniffi_checksum_constructor_slatedbwalreader_new() + }) + if checksum != 59531 { + // If this happens try cleaning and rebuilding your project + panic("slatedb: uniffi_slatedb_uniffi_checksum_constructor_slatedbwalreader_new: UniFFI API checksum mismatch") + } + } + { + checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { + return C.uniffi_slatedb_uniffi_checksum_constructor_slatedbwalreader_with_options() + }) + if checksum != 41949 { + // If this happens try cleaning and rebuilding your project + panic("slatedb: uniffi_slatedb_uniffi_checksum_constructor_slatedbwalreader_with_options: UniFFI API checksum mismatch") + } + } + { + checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { + return C.uniffi_slatedb_uniffi_checksum_constructor_slatedbwalreader_with_wal_object_store() + }) + if checksum != 35831 { + // If this happens try cleaning and rebuilding your project + panic("slatedb: uniffi_slatedb_uniffi_checksum_constructor_slatedbwalreader_with_wal_object_store: UniFFI API checksum mismatch") + } + } + { + checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { + return C.uniffi_slatedb_uniffi_checksum_constructor_slatedbwalreader_with_wal_object_store_and_options() }) - if checksum != 30537 { + if checksum != 28081 { // If this happens try cleaning and rebuilding your project - panic("slatedb: uniffi_slatedb_uniffi_checksum_constructor_walreader_new: UniFFI API checksum mismatch") + panic("slatedb: uniffi_slatedb_uniffi_checksum_constructor_slatedbwalreader_with_wal_object_store_and_options: UniFFI API checksum mismatch") } } { @@ -8246,176 +8228,181 @@ func (_ FfiDestroyerSettings) Destroy(value *Settings) { value.Destroy() } -// Handle for an up/down counter metric. -type UpDownCounter interface { - // Adds `value` to the counter. - Increment(value int64) +// Live iterator over SlateDB WAL files starting at a required WAL file ID. +type SlateDbWalIteratorInterface interface { + // Returns rows from the next fully consumed WAL file. When it reaches the + // current tail, this call waits for the next WAL file rather than ending. + Next() (*WalRows, error) } -// Handle for an up/down counter metric. -type UpDownCounterImpl struct { +// Live iterator over SlateDB WAL files starting at a required WAL file ID. +type SlateDbWalIterator struct { ffiObject FfiObject } -// Adds `value` to the counter. -func (_self *UpDownCounterImpl) Increment(value int64) { - _pointer := _self.ffiObject.incrementPointer("UpDownCounter") +// Returns rows from the next fully consumed WAL file. When it reaches the +// current tail, this call waits for the next WAL file rather than ending. +func (_self *SlateDbWalIterator) Next() (*WalRows, error) { + _pointer := _self.ffiObject.incrementPointer("*SlateDbWalIterator") defer _self.ffiObject.decrementPointer() - rustCall(func(_uniffiStatus *C.RustCallStatus) bool { - C.uniffi_slatedb_uniffi_fn_method_updowncounter_increment( - _pointer, FfiConverterInt64INSTANCE.Lower(value), _uniffiStatus) - return false - }) + res, err := uniffiRustCallAsync[*Error]( + FfiConverterErrorINSTANCE, + // completeFn + func(handle C.uint64_t, status *C.RustCallStatus) RustBufferI { + res := C.ffi_slatedb_uniffi_rust_future_complete_rust_buffer(handle, status) + return GoRustBuffer{ + inner: res, + } + }, + // liftFn + func(ffi RustBufferI) *WalRows { + return FfiConverterOptionalWalRowsINSTANCE.Lift(ffi) + }, + C.uniffi_slatedb_uniffi_fn_method_slatedbwaliterator_next( + _pointer), + // pollFn + func(handle C.uint64_t, continuation C.UniffiRustFutureContinuationCallback, data C.uint64_t) { + C.ffi_slatedb_uniffi_rust_future_poll_rust_buffer(handle, continuation, data) + }, + // freeFn + func(handle C.uint64_t) { + C.ffi_slatedb_uniffi_rust_future_free_rust_buffer(handle) + }, + ) + + if err == nil { + return res, nil + } + + return res, err } -func (object *UpDownCounterImpl) Destroy() { +func (object *SlateDbWalIterator) Destroy() { runtime.SetFinalizer(object, nil) object.ffiObject.destroy() } -type FfiConverterUpDownCounter struct { - handleMap *concurrentHandleMap[UpDownCounter] -} +type FfiConverterSlateDbWalIterator struct{} -var FfiConverterUpDownCounterINSTANCE = FfiConverterUpDownCounter{ - handleMap: newConcurrentHandleMap[UpDownCounter](), -} +var FfiConverterSlateDbWalIteratorINSTANCE = FfiConverterSlateDbWalIterator{} -func (c FfiConverterUpDownCounter) Lift(handle C.uint64_t) UpDownCounter { - if uint64(handle)&1 == 0 { - // Rust-generated handle (even), construct a new object wrapping the handle - result := &UpDownCounterImpl{ - newFfiObject( - handle, - func(handle C.uint64_t, status *C.RustCallStatus) C.uint64_t { - return C.uniffi_slatedb_uniffi_fn_clone_updowncounter(handle, status) - }, - func(handle C.uint64_t, status *C.RustCallStatus) { - C.uniffi_slatedb_uniffi_fn_free_updowncounter(handle, status) - }, - ), - } - runtime.SetFinalizer(result, (*UpDownCounterImpl).Destroy) - return result - } else { - // Go-generated handle (odd), retrieve from the handle map - val, ok := c.handleMap.tryGet(uint64(handle)) - if !ok { - panic(fmt.Errorf("no callback in handle map: %d", handle)) - } - c.handleMap.remove(uint64(handle)) - return val +func (c FfiConverterSlateDbWalIterator) Lift(handle C.uint64_t) *SlateDbWalIterator { + result := &SlateDbWalIterator{ + newFfiObject( + handle, + func(handle C.uint64_t, status *C.RustCallStatus) C.uint64_t { + return C.uniffi_slatedb_uniffi_fn_clone_slatedbwaliterator(handle, status) + }, + func(handle C.uint64_t, status *C.RustCallStatus) { + C.uniffi_slatedb_uniffi_fn_free_slatedbwaliterator(handle, status) + }, + ), } + runtime.SetFinalizer(result, (*SlateDbWalIterator).Destroy) + return result } -func (c FfiConverterUpDownCounter) Read(reader io.Reader) UpDownCounter { +func (c FfiConverterSlateDbWalIterator) Read(reader io.Reader) *SlateDbWalIterator { return c.Lift(C.uint64_t(readUint64(reader))) } -func (c FfiConverterUpDownCounter) Lower(value UpDownCounter) C.uint64_t { +func (c FfiConverterSlateDbWalIterator) Lower(value *SlateDbWalIterator) C.uint64_t { // TODO: this is bad - all synchronization from ObjectRuntime.go is discarded here, // because the handle will be decremented immediately after this function returns, // and someone will be left holding onto a non-locked handle. - if val, ok := value.(*UpDownCounterImpl); ok { - // Rust-backed object, clone the handle - handle := val.ffiObject.incrementPointer("UpDownCounter") - defer val.ffiObject.decrementPointer() - return handle - } else { - // Go-backed object, insert into handle map - return C.uint64_t(c.handleMap.insert(value)) - } + handle := value.ffiObject.incrementPointer("*SlateDbWalIterator") + defer value.ffiObject.decrementPointer() + return handle } -func (c FfiConverterUpDownCounter) Write(writer io.Writer, value UpDownCounter) { +func (c FfiConverterSlateDbWalIterator) Write(writer io.Writer, value *SlateDbWalIterator) { writeUint64(writer, uint64(c.Lower(value))) } -func LiftFromExternalUpDownCounter(handle uint64) UpDownCounter { - return FfiConverterUpDownCounterINSTANCE.Lift(C.uint64_t(handle)) -} - -func LowerToExternalUpDownCounter(value UpDownCounter) uint64 { - return uint64(FfiConverterUpDownCounterINSTANCE.Lower(value)) +func LiftFromExternalSlateDbWalIterator(handle uint64) *SlateDbWalIterator { + return FfiConverterSlateDbWalIteratorINSTANCE.Lift(C.uint64_t(handle)) } -type FfiDestroyerUpDownCounter struct{} - -func (_ FfiDestroyerUpDownCounter) Destroy(value UpDownCounter) { - if val, ok := value.(*UpDownCounterImpl); ok { - val.Destroy() - } +func LowerToExternalSlateDbWalIterator(value *SlateDbWalIterator) uint64 { + return uint64(FfiConverterSlateDbWalIteratorINSTANCE.Lower(value)) } -//export slatedb_uniffi_metrics_cgo_dispatchCallbackInterfaceUpDownCounterMethod0 -func slatedb_uniffi_metrics_cgo_dispatchCallbackInterfaceUpDownCounterMethod0(uniffiHandle C.uint64_t, value C.int64_t, uniffiOutReturn *C.void, callStatus *C.RustCallStatus) { - handle := uint64(uniffiHandle) - uniffiObj, ok := FfiConverterUpDownCounterINSTANCE.handleMap.tryGet(handle) - if !ok { - panic(fmt.Errorf("no callback in handle map: %d", handle)) - } - - uniffiObj.Increment( - FfiConverterInt64INSTANCE.Lift(value), - ) +type FfiDestroyerSlateDbWalIterator struct{} +func (_ FfiDestroyerSlateDbWalIterator) Destroy(value *SlateDbWalIterator) { + value.Destroy() } -var UniffiVTableCallbackInterfaceUpDownCounterINSTANCE = C.UniffiVTableCallbackInterfaceUpDownCounter{ - uniffiFree: (C.UniffiCallbackInterfaceFree)(C.slatedb_uniffi_metrics_cgo_dispatchCallbackInterfaceUpDownCounterFree), - uniffiClone: (C.UniffiCallbackInterfaceClone)(C.slatedb_uniffi_metrics_cgo_dispatchCallbackInterfaceUpDownCounterClone), - increment: (C.UniffiCallbackInterfaceUpDownCounterMethod0)(C.slatedb_uniffi_metrics_cgo_dispatchCallbackInterfaceUpDownCounterMethod0), +// CDC reader backed by SlateDB's native live WAL reader. +type SlateDbWalReaderInterface interface { + // Opens a live iterator starting at start_wal_file_id. The iterator waits + // and polls internally when it reaches the current WAL tail. + Iterator(startWalFileId uint64) (*SlateDbWalIterator, error) + // Returns a snapshot of the current WAL tail after replay_after_wal_id, or + // the supplied ID when no later WAL file exists. + LastWalFileId(replayAfterWalId uint64) (uint64, error) } -//export slatedb_uniffi_metrics_cgo_dispatchCallbackInterfaceUpDownCounterFree -func slatedb_uniffi_metrics_cgo_dispatchCallbackInterfaceUpDownCounterFree(handle C.uint64_t) { - FfiConverterUpDownCounterINSTANCE.handleMap.remove(uint64(handle)) +// CDC reader backed by SlateDB's native live WAL reader. +type SlateDbWalReader struct { + ffiObject FfiObject } -//export slatedb_uniffi_metrics_cgo_dispatchCallbackInterfaceUpDownCounterClone -func slatedb_uniffi_metrics_cgo_dispatchCallbackInterfaceUpDownCounterClone(handle C.uint64_t) C.uint64_t { - val, ok := FfiConverterUpDownCounterINSTANCE.handleMap.tryGet(uint64(handle)) - if !ok { - panic(fmt.Errorf("no callback in handle map: %d", handle)) +// Opens a reader when the manifest and WAL use the same object store. +func NewSlateDbWalReader(path string, objectStore *ObjectStore) (*SlateDbWalReader, error) { + _uniffiRV, _uniffiErr := rustCallWithError[*Error](FfiConverterError{}, func(_uniffiStatus *C.RustCallStatus) C.uint64_t { + return C.uniffi_slatedb_uniffi_fn_constructor_slatedbwalreader_new(FfiConverterStringINSTANCE.Lower(path), FfiConverterObjectStoreINSTANCE.Lower(objectStore), _uniffiStatus) + }) + if _uniffiErr != nil { + var _uniffiDefaultValue *SlateDbWalReader + return _uniffiDefaultValue, _uniffiErr + } else { + return FfiConverterSlateDbWalReaderINSTANCE.Lift(_uniffiRV), nil } - return C.uint64_t(FfiConverterUpDownCounterINSTANCE.handleMap.insert(val)) -} - -func (c FfiConverterUpDownCounter) register() { - C.uniffi_slatedb_uniffi_fn_init_callback_vtable_updowncounter(&UniffiVTableCallbackInterfaceUpDownCounterINSTANCE) } -// Handle for a single WAL file. -type WalFileInterface interface { - // Returns the WAL file ID. - Id() uint64 - // Opens an iterator over raw row entries in this WAL file. - Iterator() (*WalFileIterator, error) - // Reads object-store metadata for this WAL file. - Metadata() (IdentifiedObjectMetadata, error) - // Returns a handle for the next WAL file ID without checking existence. - NextFile() *WalFile - // Returns the WAL ID immediately after this file. - NextId() uint64 +// Opens a reader with explicit fetch options. +func SlateDbWalReaderWithOptions(path string, objectStore *ObjectStore, options SlateDbWalReaderOptions) (*SlateDbWalReader, error) { + _uniffiRV, _uniffiErr := rustCallWithError[*Error](FfiConverterError{}, func(_uniffiStatus *C.RustCallStatus) C.uint64_t { + return C.uniffi_slatedb_uniffi_fn_constructor_slatedbwalreader_with_options(FfiConverterStringINSTANCE.Lower(path), FfiConverterObjectStoreINSTANCE.Lower(objectStore), FfiConverterSlateDbWalReaderOptionsINSTANCE.Lower(options), _uniffiStatus) + }) + if _uniffiErr != nil { + var _uniffiDefaultValue *SlateDbWalReader + return _uniffiDefaultValue, _uniffiErr + } else { + return FfiConverterSlateDbWalReaderINSTANCE.Lift(_uniffiRV), nil + } } -// Handle for a single WAL file. -type WalFile struct { - ffiObject FfiObject +// Opens a reader for a database with a dedicated WAL object store. +func SlateDbWalReaderWithWalObjectStore(path string, objectStore *ObjectStore, walObjectStore *ObjectStore) (*SlateDbWalReader, error) { + _uniffiRV, _uniffiErr := rustCallWithError[*Error](FfiConverterError{}, func(_uniffiStatus *C.RustCallStatus) C.uint64_t { + return C.uniffi_slatedb_uniffi_fn_constructor_slatedbwalreader_with_wal_object_store(FfiConverterStringINSTANCE.Lower(path), FfiConverterObjectStoreINSTANCE.Lower(objectStore), FfiConverterObjectStoreINSTANCE.Lower(walObjectStore), _uniffiStatus) + }) + if _uniffiErr != nil { + var _uniffiDefaultValue *SlateDbWalReader + return _uniffiDefaultValue, _uniffiErr + } else { + return FfiConverterSlateDbWalReaderINSTANCE.Lift(_uniffiRV), nil + } } -// Returns the WAL file ID. -func (_self *WalFile) Id() uint64 { - _pointer := _self.ffiObject.incrementPointer("*WalFile") - defer _self.ffiObject.decrementPointer() - return FfiConverterUint64INSTANCE.Lift(rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint64_t { - return C.uniffi_slatedb_uniffi_fn_method_walfile_id( - _pointer, _uniffiStatus) - })) +// Opens a reader for a dedicated WAL object store with explicit options. +func SlateDbWalReaderWithWalObjectStoreAndOptions(path string, objectStore *ObjectStore, walObjectStore *ObjectStore, options SlateDbWalReaderOptions) (*SlateDbWalReader, error) { + _uniffiRV, _uniffiErr := rustCallWithError[*Error](FfiConverterError{}, func(_uniffiStatus *C.RustCallStatus) C.uint64_t { + return C.uniffi_slatedb_uniffi_fn_constructor_slatedbwalreader_with_wal_object_store_and_options(FfiConverterStringINSTANCE.Lower(path), FfiConverterObjectStoreINSTANCE.Lower(objectStore), FfiConverterObjectStoreINSTANCE.Lower(walObjectStore), FfiConverterSlateDbWalReaderOptionsINSTANCE.Lower(options), _uniffiStatus) + }) + if _uniffiErr != nil { + var _uniffiDefaultValue *SlateDbWalReader + return _uniffiDefaultValue, _uniffiErr + } else { + return FfiConverterSlateDbWalReaderINSTANCE.Lift(_uniffiRV), nil + } } -// Opens an iterator over raw row entries in this WAL file. -func (_self *WalFile) Iterator() (*WalFileIterator, error) { - _pointer := _self.ffiObject.incrementPointer("*WalFile") +// Opens a live iterator starting at start_wal_file_id. The iterator waits +// and polls internally when it reaches the current WAL tail. +func (_self *SlateDbWalReader) Iterator(startWalFileId uint64) (*SlateDbWalIterator, error) { + _pointer := _self.ffiObject.incrementPointer("*SlateDbWalReader") defer _self.ffiObject.decrementPointer() res, err := uniffiRustCallAsync[*Error]( FfiConverterErrorINSTANCE, @@ -8425,11 +8412,11 @@ func (_self *WalFile) Iterator() (*WalFileIterator, error) { return res }, // liftFn - func(ffi C.uint64_t) *WalFileIterator { - return FfiConverterWalFileIteratorINSTANCE.Lift(ffi) + func(ffi C.uint64_t) *SlateDbWalIterator { + return FfiConverterSlateDbWalIteratorINSTANCE.Lift(ffi) }, - C.uniffi_slatedb_uniffi_fn_method_walfile_iterator( - _pointer), + C.uniffi_slatedb_uniffi_fn_method_slatedbwalreader_iterator( + _pointer, FfiConverterUint64INSTANCE.Lower(startWalFileId)), // pollFn func(handle C.uint64_t, continuation C.UniffiRustFutureContinuationCallback, data C.uint64_t) { C.ffi_slatedb_uniffi_rust_future_poll_u64(handle, continuation, data) @@ -8447,32 +8434,31 @@ func (_self *WalFile) Iterator() (*WalFileIterator, error) { return res, err } -// Reads object-store metadata for this WAL file. -func (_self *WalFile) Metadata() (IdentifiedObjectMetadata, error) { - _pointer := _self.ffiObject.incrementPointer("*WalFile") +// Returns a snapshot of the current WAL tail after replay_after_wal_id, or +// the supplied ID when no later WAL file exists. +func (_self *SlateDbWalReader) LastWalFileId(replayAfterWalId uint64) (uint64, error) { + _pointer := _self.ffiObject.incrementPointer("*SlateDbWalReader") defer _self.ffiObject.decrementPointer() res, err := uniffiRustCallAsync[*Error]( FfiConverterErrorINSTANCE, // completeFn - func(handle C.uint64_t, status *C.RustCallStatus) RustBufferI { - res := C.ffi_slatedb_uniffi_rust_future_complete_rust_buffer(handle, status) - return GoRustBuffer{ - inner: res, - } + func(handle C.uint64_t, status *C.RustCallStatus) C.uint64_t { + res := C.ffi_slatedb_uniffi_rust_future_complete_u64(handle, status) + return res }, // liftFn - func(ffi RustBufferI) IdentifiedObjectMetadata { - return FfiConverterIdentifiedObjectMetadataINSTANCE.Lift(ffi) + func(ffi C.uint64_t) uint64 { + return FfiConverterUint64INSTANCE.Lift(ffi) }, - C.uniffi_slatedb_uniffi_fn_method_walfile_metadata( - _pointer), + C.uniffi_slatedb_uniffi_fn_method_slatedbwalreader_last_wal_file_id( + _pointer, FfiConverterUint64INSTANCE.Lower(replayAfterWalId)), // pollFn func(handle C.uint64_t, continuation C.UniffiRustFutureContinuationCallback, data C.uint64_t) { - C.ffi_slatedb_uniffi_rust_future_poll_rust_buffer(handle, continuation, data) + C.ffi_slatedb_uniffi_rust_future_poll_u64(handle, continuation, data) }, // freeFn func(handle C.uint64_t) { - C.ffi_slatedb_uniffi_rust_future_free_rust_buffer(handle) + C.ffi_slatedb_uniffi_rust_future_free_u64(handle) }, ) @@ -8482,307 +8468,198 @@ func (_self *WalFile) Metadata() (IdentifiedObjectMetadata, error) { return res, err } - -// Returns a handle for the next WAL file ID without checking existence. -func (_self *WalFile) NextFile() *WalFile { - _pointer := _self.ffiObject.incrementPointer("*WalFile") - defer _self.ffiObject.decrementPointer() - return FfiConverterWalFileINSTANCE.Lift(rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint64_t { - return C.uniffi_slatedb_uniffi_fn_method_walfile_next_file( - _pointer, _uniffiStatus) - })) -} - -// Returns the WAL ID immediately after this file. -func (_self *WalFile) NextId() uint64 { - _pointer := _self.ffiObject.incrementPointer("*WalFile") - defer _self.ffiObject.decrementPointer() - return FfiConverterUint64INSTANCE.Lift(rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint64_t { - return C.uniffi_slatedb_uniffi_fn_method_walfile_next_id( - _pointer, _uniffiStatus) - })) -} -func (object *WalFile) Destroy() { +func (object *SlateDbWalReader) Destroy() { runtime.SetFinalizer(object, nil) object.ffiObject.destroy() } -type FfiConverterWalFile struct{} +type FfiConverterSlateDbWalReader struct{} -var FfiConverterWalFileINSTANCE = FfiConverterWalFile{} +var FfiConverterSlateDbWalReaderINSTANCE = FfiConverterSlateDbWalReader{} -func (c FfiConverterWalFile) Lift(handle C.uint64_t) *WalFile { - result := &WalFile{ +func (c FfiConverterSlateDbWalReader) Lift(handle C.uint64_t) *SlateDbWalReader { + result := &SlateDbWalReader{ newFfiObject( handle, func(handle C.uint64_t, status *C.RustCallStatus) C.uint64_t { - return C.uniffi_slatedb_uniffi_fn_clone_walfile(handle, status) + return C.uniffi_slatedb_uniffi_fn_clone_slatedbwalreader(handle, status) }, func(handle C.uint64_t, status *C.RustCallStatus) { - C.uniffi_slatedb_uniffi_fn_free_walfile(handle, status) + C.uniffi_slatedb_uniffi_fn_free_slatedbwalreader(handle, status) }, ), } - runtime.SetFinalizer(result, (*WalFile).Destroy) + runtime.SetFinalizer(result, (*SlateDbWalReader).Destroy) return result } -func (c FfiConverterWalFile) Read(reader io.Reader) *WalFile { +func (c FfiConverterSlateDbWalReader) Read(reader io.Reader) *SlateDbWalReader { return c.Lift(C.uint64_t(readUint64(reader))) } -func (c FfiConverterWalFile) Lower(value *WalFile) C.uint64_t { +func (c FfiConverterSlateDbWalReader) Lower(value *SlateDbWalReader) C.uint64_t { // TODO: this is bad - all synchronization from ObjectRuntime.go is discarded here, // because the handle will be decremented immediately after this function returns, // and someone will be left holding onto a non-locked handle. - handle := value.ffiObject.incrementPointer("*WalFile") + handle := value.ffiObject.incrementPointer("*SlateDbWalReader") defer value.ffiObject.decrementPointer() return handle } -func (c FfiConverterWalFile) Write(writer io.Writer, value *WalFile) { +func (c FfiConverterSlateDbWalReader) Write(writer io.Writer, value *SlateDbWalReader) { writeUint64(writer, uint64(c.Lower(value))) } -func LiftFromExternalWalFile(handle uint64) *WalFile { - return FfiConverterWalFileINSTANCE.Lift(C.uint64_t(handle)) +func LiftFromExternalSlateDbWalReader(handle uint64) *SlateDbWalReader { + return FfiConverterSlateDbWalReaderINSTANCE.Lift(C.uint64_t(handle)) } -func LowerToExternalWalFile(value *WalFile) uint64 { - return uint64(FfiConverterWalFileINSTANCE.Lower(value)) +func LowerToExternalSlateDbWalReader(value *SlateDbWalReader) uint64 { + return uint64(FfiConverterSlateDbWalReaderINSTANCE.Lower(value)) } -type FfiDestroyerWalFile struct{} +type FfiDestroyerSlateDbWalReader struct{} -func (_ FfiDestroyerWalFile) Destroy(value *WalFile) { +func (_ FfiDestroyerSlateDbWalReader) Destroy(value *SlateDbWalReader) { value.Destroy() } -// Iterator over raw row entries stored in a WAL file. -type WalFileIteratorInterface interface { - // Returns the next raw row entry from the WAL file. - Next() (*RowEntry, error) +// Handle for an up/down counter metric. +type UpDownCounter interface { + // Adds `value` to the counter. + Increment(value int64) } -// Iterator over raw row entries stored in a WAL file. -type WalFileIterator struct { +// Handle for an up/down counter metric. +type UpDownCounterImpl struct { ffiObject FfiObject } -// Returns the next raw row entry from the WAL file. -func (_self *WalFileIterator) Next() (*RowEntry, error) { - _pointer := _self.ffiObject.incrementPointer("*WalFileIterator") +// Adds `value` to the counter. +func (_self *UpDownCounterImpl) Increment(value int64) { + _pointer := _self.ffiObject.incrementPointer("UpDownCounter") defer _self.ffiObject.decrementPointer() - res, err := uniffiRustCallAsync[*Error]( - FfiConverterErrorINSTANCE, - // completeFn - func(handle C.uint64_t, status *C.RustCallStatus) RustBufferI { - res := C.ffi_slatedb_uniffi_rust_future_complete_rust_buffer(handle, status) - return GoRustBuffer{ - inner: res, - } - }, - // liftFn - func(ffi RustBufferI) *RowEntry { - return FfiConverterOptionalRowEntryINSTANCE.Lift(ffi) - }, - C.uniffi_slatedb_uniffi_fn_method_walfileiterator_next( - _pointer), - // pollFn - func(handle C.uint64_t, continuation C.UniffiRustFutureContinuationCallback, data C.uint64_t) { - C.ffi_slatedb_uniffi_rust_future_poll_rust_buffer(handle, continuation, data) - }, - // freeFn - func(handle C.uint64_t) { - C.ffi_slatedb_uniffi_rust_future_free_rust_buffer(handle) - }, - ) - - if err == nil { - return res, nil - } - - return res, err + rustCall(func(_uniffiStatus *C.RustCallStatus) bool { + C.uniffi_slatedb_uniffi_fn_method_updowncounter_increment( + _pointer, FfiConverterInt64INSTANCE.Lower(value), _uniffiStatus) + return false + }) } -func (object *WalFileIterator) Destroy() { +func (object *UpDownCounterImpl) Destroy() { runtime.SetFinalizer(object, nil) object.ffiObject.destroy() } -type FfiConverterWalFileIterator struct{} +type FfiConverterUpDownCounter struct { + handleMap *concurrentHandleMap[UpDownCounter] +} -var FfiConverterWalFileIteratorINSTANCE = FfiConverterWalFileIterator{} +var FfiConverterUpDownCounterINSTANCE = FfiConverterUpDownCounter{ + handleMap: newConcurrentHandleMap[UpDownCounter](), +} -func (c FfiConverterWalFileIterator) Lift(handle C.uint64_t) *WalFileIterator { - result := &WalFileIterator{ - newFfiObject( - handle, - func(handle C.uint64_t, status *C.RustCallStatus) C.uint64_t { - return C.uniffi_slatedb_uniffi_fn_clone_walfileiterator(handle, status) - }, - func(handle C.uint64_t, status *C.RustCallStatus) { - C.uniffi_slatedb_uniffi_fn_free_walfileiterator(handle, status) - }, - ), +func (c FfiConverterUpDownCounter) Lift(handle C.uint64_t) UpDownCounter { + if uint64(handle)&1 == 0 { + // Rust-generated handle (even), construct a new object wrapping the handle + result := &UpDownCounterImpl{ + newFfiObject( + handle, + func(handle C.uint64_t, status *C.RustCallStatus) C.uint64_t { + return C.uniffi_slatedb_uniffi_fn_clone_updowncounter(handle, status) + }, + func(handle C.uint64_t, status *C.RustCallStatus) { + C.uniffi_slatedb_uniffi_fn_free_updowncounter(handle, status) + }, + ), + } + runtime.SetFinalizer(result, (*UpDownCounterImpl).Destroy) + return result + } else { + // Go-generated handle (odd), retrieve from the handle map + val, ok := c.handleMap.tryGet(uint64(handle)) + if !ok { + panic(fmt.Errorf("no callback in handle map: %d", handle)) + } + c.handleMap.remove(uint64(handle)) + return val } - runtime.SetFinalizer(result, (*WalFileIterator).Destroy) - return result } -func (c FfiConverterWalFileIterator) Read(reader io.Reader) *WalFileIterator { +func (c FfiConverterUpDownCounter) Read(reader io.Reader) UpDownCounter { return c.Lift(C.uint64_t(readUint64(reader))) } -func (c FfiConverterWalFileIterator) Lower(value *WalFileIterator) C.uint64_t { +func (c FfiConverterUpDownCounter) Lower(value UpDownCounter) C.uint64_t { // TODO: this is bad - all synchronization from ObjectRuntime.go is discarded here, // because the handle will be decremented immediately after this function returns, // and someone will be left holding onto a non-locked handle. - handle := value.ffiObject.incrementPointer("*WalFileIterator") - defer value.ffiObject.decrementPointer() - return handle + if val, ok := value.(*UpDownCounterImpl); ok { + // Rust-backed object, clone the handle + handle := val.ffiObject.incrementPointer("UpDownCounter") + defer val.ffiObject.decrementPointer() + return handle + } else { + // Go-backed object, insert into handle map + return C.uint64_t(c.handleMap.insert(value)) + } } -func (c FfiConverterWalFileIterator) Write(writer io.Writer, value *WalFileIterator) { +func (c FfiConverterUpDownCounter) Write(writer io.Writer, value UpDownCounter) { writeUint64(writer, uint64(c.Lower(value))) } -func LiftFromExternalWalFileIterator(handle uint64) *WalFileIterator { - return FfiConverterWalFileIteratorINSTANCE.Lift(C.uint64_t(handle)) -} - -func LowerToExternalWalFileIterator(value *WalFileIterator) uint64 { - return uint64(FfiConverterWalFileIteratorINSTANCE.Lower(value)) -} - -type FfiDestroyerWalFileIterator struct{} - -func (_ FfiDestroyerWalFileIterator) Destroy(value *WalFileIterator) { - value.Destroy() -} - -// Reader for WAL files stored under a database path. -type WalReaderInterface interface { - // Returns a handle for the WAL file with the given ID. - Get(id uint64) *WalFile - // Lists WAL files in ascending ID order. - // - // `start_id` is inclusive and `end_id` is exclusive when provided. - List(startId *uint64, endId *uint64) ([]*WalFile, error) -} - -// Reader for WAL files stored under a database path. -type WalReader struct { - ffiObject FfiObject -} - -// Creates a WAL reader for `path` in `object_store`. -func NewWalReader(path string, objectStore *ObjectStore) *WalReader { - return FfiConverterWalReaderINSTANCE.Lift(rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint64_t { - return C.uniffi_slatedb_uniffi_fn_constructor_walreader_new(FfiConverterStringINSTANCE.Lower(path), FfiConverterObjectStoreINSTANCE.Lower(objectStore), _uniffiStatus) - })) +func LiftFromExternalUpDownCounter(handle uint64) UpDownCounter { + return FfiConverterUpDownCounterINSTANCE.Lift(C.uint64_t(handle)) } -// Returns a handle for the WAL file with the given ID. -func (_self *WalReader) Get(id uint64) *WalFile { - _pointer := _self.ffiObject.incrementPointer("*WalReader") - defer _self.ffiObject.decrementPointer() - return FfiConverterWalFileINSTANCE.Lift(rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint64_t { - return C.uniffi_slatedb_uniffi_fn_method_walreader_get( - _pointer, FfiConverterUint64INSTANCE.Lower(id), _uniffiStatus) - })) +func LowerToExternalUpDownCounter(value UpDownCounter) uint64 { + return uint64(FfiConverterUpDownCounterINSTANCE.Lower(value)) } -// Lists WAL files in ascending ID order. -// -// `start_id` is inclusive and `end_id` is exclusive when provided. -func (_self *WalReader) List(startId *uint64, endId *uint64) ([]*WalFile, error) { - _pointer := _self.ffiObject.incrementPointer("*WalReader") - defer _self.ffiObject.decrementPointer() - res, err := uniffiRustCallAsync[*Error]( - FfiConverterErrorINSTANCE, - // completeFn - func(handle C.uint64_t, status *C.RustCallStatus) RustBufferI { - res := C.ffi_slatedb_uniffi_rust_future_complete_rust_buffer(handle, status) - return GoRustBuffer{ - inner: res, - } - }, - // liftFn - func(ffi RustBufferI) []*WalFile { - return FfiConverterSequenceWalFileINSTANCE.Lift(ffi) - }, - C.uniffi_slatedb_uniffi_fn_method_walreader_list( - _pointer, FfiConverterOptionalUint64INSTANCE.Lower(startId), FfiConverterOptionalUint64INSTANCE.Lower(endId)), - // pollFn - func(handle C.uint64_t, continuation C.UniffiRustFutureContinuationCallback, data C.uint64_t) { - C.ffi_slatedb_uniffi_rust_future_poll_rust_buffer(handle, continuation, data) - }, - // freeFn - func(handle C.uint64_t) { - C.ffi_slatedb_uniffi_rust_future_free_rust_buffer(handle) - }, - ) +type FfiDestroyerUpDownCounter struct{} - if err == nil { - return res, nil +func (_ FfiDestroyerUpDownCounter) Destroy(value UpDownCounter) { + if val, ok := value.(*UpDownCounterImpl); ok { + val.Destroy() } - - return res, err } -func (object *WalReader) Destroy() { - runtime.SetFinalizer(object, nil) - object.ffiObject.destroy() -} - -type FfiConverterWalReader struct{} - -var FfiConverterWalReaderINSTANCE = FfiConverterWalReader{} -func (c FfiConverterWalReader) Lift(handle C.uint64_t) *WalReader { - result := &WalReader{ - newFfiObject( - handle, - func(handle C.uint64_t, status *C.RustCallStatus) C.uint64_t { - return C.uniffi_slatedb_uniffi_fn_clone_walreader(handle, status) - }, - func(handle C.uint64_t, status *C.RustCallStatus) { - C.uniffi_slatedb_uniffi_fn_free_walreader(handle, status) - }, - ), +//export slatedb_uniffi_metrics_cgo_dispatchCallbackInterfaceUpDownCounterMethod0 +func slatedb_uniffi_metrics_cgo_dispatchCallbackInterfaceUpDownCounterMethod0(uniffiHandle C.uint64_t, value C.int64_t, uniffiOutReturn *C.void, callStatus *C.RustCallStatus) { + handle := uint64(uniffiHandle) + uniffiObj, ok := FfiConverterUpDownCounterINSTANCE.handleMap.tryGet(handle) + if !ok { + panic(fmt.Errorf("no callback in handle map: %d", handle)) } - runtime.SetFinalizer(result, (*WalReader).Destroy) - return result -} -func (c FfiConverterWalReader) Read(reader io.Reader) *WalReader { - return c.Lift(C.uint64_t(readUint64(reader))) -} + uniffiObj.Increment( + FfiConverterInt64INSTANCE.Lift(value), + ) -func (c FfiConverterWalReader) Lower(value *WalReader) C.uint64_t { - // TODO: this is bad - all synchronization from ObjectRuntime.go is discarded here, - // because the handle will be decremented immediately after this function returns, - // and someone will be left holding onto a non-locked handle. - handle := value.ffiObject.incrementPointer("*WalReader") - defer value.ffiObject.decrementPointer() - return handle } -func (c FfiConverterWalReader) Write(writer io.Writer, value *WalReader) { - writeUint64(writer, uint64(c.Lower(value))) +var UniffiVTableCallbackInterfaceUpDownCounterINSTANCE = C.UniffiVTableCallbackInterfaceUpDownCounter{ + uniffiFree: (C.UniffiCallbackInterfaceFree)(C.slatedb_uniffi_metrics_cgo_dispatchCallbackInterfaceUpDownCounterFree), + uniffiClone: (C.UniffiCallbackInterfaceClone)(C.slatedb_uniffi_metrics_cgo_dispatchCallbackInterfaceUpDownCounterClone), + increment: (C.UniffiCallbackInterfaceUpDownCounterMethod0)(C.slatedb_uniffi_metrics_cgo_dispatchCallbackInterfaceUpDownCounterMethod0), } -func LiftFromExternalWalReader(handle uint64) *WalReader { - return FfiConverterWalReaderINSTANCE.Lift(C.uint64_t(handle)) +//export slatedb_uniffi_metrics_cgo_dispatchCallbackInterfaceUpDownCounterFree +func slatedb_uniffi_metrics_cgo_dispatchCallbackInterfaceUpDownCounterFree(handle C.uint64_t) { + FfiConverterUpDownCounterINSTANCE.handleMap.remove(uint64(handle)) } -func LowerToExternalWalReader(value *WalReader) uint64 { - return uint64(FfiConverterWalReaderINSTANCE.Lower(value)) +//export slatedb_uniffi_metrics_cgo_dispatchCallbackInterfaceUpDownCounterClone +func slatedb_uniffi_metrics_cgo_dispatchCallbackInterfaceUpDownCounterClone(handle C.uint64_t) C.uint64_t { + val, ok := FfiConverterUpDownCounterINSTANCE.handleMap.tryGet(uint64(handle)) + if !ok { + panic(fmt.Errorf("no callback in handle map: %d", handle)) + } + return C.uint64_t(FfiConverterUpDownCounterINSTANCE.handleMap.insert(val)) } -type FfiDestroyerWalReader struct{} - -func (_ FfiDestroyerWalReader) Destroy(value *WalReader) { - value.Destroy() +func (c FfiConverterUpDownCounter) register() { + C.uniffi_slatedb_uniffi_fn_init_callback_vtable_updowncounter(&UniffiVTableCallbackInterfaceUpDownCounterINSTANCE) } // Mutable batch of write operations applied atomically by [`crate::Db::write`]. @@ -10831,6 +10708,58 @@ func (_ FfiDestroyerSegmentPrefix) Destroy(value SegmentPrefix) { value.Destroy() } +// Options controlling how the native SlateDB WAL reader fetches WAL SSTs. +type SlateDbWalReaderOptions struct { + // Number of WAL SSTs to preload. + SstBatchSize uint64 + // Number of concurrent fetch tasks per WAL SST. + MaxFetchTasks uint64 + // Number of bytes to read ahead from each WAL SST. + ReadAheadBytes uint64 +} + +func (r *SlateDbWalReaderOptions) Destroy() { + FfiDestroyerUint64{}.Destroy(r.SstBatchSize) + FfiDestroyerUint64{}.Destroy(r.MaxFetchTasks) + FfiDestroyerUint64{}.Destroy(r.ReadAheadBytes) +} + +type FfiConverterSlateDbWalReaderOptions struct{} + +var FfiConverterSlateDbWalReaderOptionsINSTANCE = FfiConverterSlateDbWalReaderOptions{} + +func (c FfiConverterSlateDbWalReaderOptions) Lift(rb RustBufferI) SlateDbWalReaderOptions { + return LiftFromRustBuffer[SlateDbWalReaderOptions](c, rb) +} + +func (c FfiConverterSlateDbWalReaderOptions) Read(reader io.Reader) SlateDbWalReaderOptions { + return SlateDbWalReaderOptions{ + FfiConverterUint64INSTANCE.Read(reader), + FfiConverterUint64INSTANCE.Read(reader), + FfiConverterUint64INSTANCE.Read(reader), + } +} + +func (c FfiConverterSlateDbWalReaderOptions) Lower(value SlateDbWalReaderOptions) C.RustBuffer { + return LowerIntoRustBuffer[SlateDbWalReaderOptions](c, value) +} + +func (c FfiConverterSlateDbWalReaderOptions) LowerExternal(value SlateDbWalReaderOptions) ExternalCRustBuffer { + return RustBufferFromC(LowerIntoRustBuffer[SlateDbWalReaderOptions](c, value)) +} + +func (c FfiConverterSlateDbWalReaderOptions) Write(writer io.Writer, value SlateDbWalReaderOptions) { + FfiConverterUint64INSTANCE.Write(writer, value.SstBatchSize) + FfiConverterUint64INSTANCE.Write(writer, value.MaxFetchTasks) + FfiConverterUint64INSTANCE.Write(writer, value.ReadAheadBytes) +} + +type FfiDestroyerSlateDbWalReaderOptions struct{} + +func (_ FfiDestroyerSlateDbWalReaderOptions) Destroy(value SlateDbWalReaderOptions) { + value.Destroy() +} + // A sorted run made up of one or more SST views. type SortedRun struct { // Sorted run ID. @@ -11258,6 +11187,53 @@ func (_ FfiDestroyerVersionedManifest) Destroy(value VersionedManifest) { value.Destroy() } +// Rows from one fully consumed WAL file. +type WalRows struct { + // Rows stored in the WAL file. Empty fence WALs produce an empty vector. + Rows []RowEntry + // Last WAL file ID fully consumed by this batch. + LastConsumedWalFileId uint64 +} + +func (r *WalRows) Destroy() { + FfiDestroyerSequenceRowEntry{}.Destroy(r.Rows) + FfiDestroyerUint64{}.Destroy(r.LastConsumedWalFileId) +} + +type FfiConverterWalRows struct{} + +var FfiConverterWalRowsINSTANCE = FfiConverterWalRows{} + +func (c FfiConverterWalRows) Lift(rb RustBufferI) WalRows { + return LiftFromRustBuffer[WalRows](c, rb) +} + +func (c FfiConverterWalRows) Read(reader io.Reader) WalRows { + return WalRows{ + FfiConverterSequenceRowEntryINSTANCE.Read(reader), + FfiConverterUint64INSTANCE.Read(reader), + } +} + +func (c FfiConverterWalRows) Lower(value WalRows) C.RustBuffer { + return LowerIntoRustBuffer[WalRows](c, value) +} + +func (c FfiConverterWalRows) LowerExternal(value WalRows) ExternalCRustBuffer { + return RustBufferFromC(LowerIntoRustBuffer[WalRows](c, value)) +} + +func (c FfiConverterWalRows) Write(writer io.Writer, value WalRows) { + FfiConverterSequenceRowEntryINSTANCE.Write(writer, value.Rows) + FfiConverterUint64INSTANCE.Write(writer, value.LastConsumedWalFileId) +} + +type FfiDestroyerWalRows struct{} + +func (_ FfiDestroyerWalRows) Destroy(value WalRows) { + value.Destroy() +} + // Options that control writes and commits. type WriteOptions struct { // Optional caller-supplied sequence number. Zero uses SlateDB's sequence oracle. @@ -13578,47 +13554,6 @@ func (_ FfiDestroyerOptionalMetric) Destroy(value *Metric) { } } -type FfiConverterOptionalRowEntry struct{} - -var FfiConverterOptionalRowEntryINSTANCE = FfiConverterOptionalRowEntry{} - -func (c FfiConverterOptionalRowEntry) Lift(rb RustBufferI) *RowEntry { - return LiftFromRustBuffer[*RowEntry](c, rb) -} - -func (_ FfiConverterOptionalRowEntry) Read(reader io.Reader) *RowEntry { - if readInt8(reader) == 0 { - return nil - } - temp := FfiConverterRowEntryINSTANCE.Read(reader) - return &temp -} - -func (c FfiConverterOptionalRowEntry) Lower(value *RowEntry) C.RustBuffer { - return LowerIntoRustBuffer[*RowEntry](c, value) -} - -func (c FfiConverterOptionalRowEntry) LowerExternal(value *RowEntry) ExternalCRustBuffer { - return RustBufferFromC(LowerIntoRustBuffer[*RowEntry](c, value)) -} - -func (_ FfiConverterOptionalRowEntry) Write(writer io.Writer, value *RowEntry) { - if value == nil { - writeInt8(writer, 0) - } else { - writeInt8(writer, 1) - FfiConverterRowEntryINSTANCE.Write(writer, *value) - } -} - -type FfiDestroyerOptionalRowEntry struct{} - -func (_ FfiDestroyerOptionalRowEntry) Destroy(value *RowEntry) { - if value != nil { - FfiDestroyerRowEntry{}.Destroy(*value) - } -} - type FfiConverterOptionalVersionedCompactions struct{} var FfiConverterOptionalVersionedCompactionsINSTANCE = FfiConverterOptionalVersionedCompactions{} @@ -13701,6 +13636,47 @@ func (_ FfiDestroyerOptionalVersionedManifest) Destroy(value *VersionedManifest) } } +type FfiConverterOptionalWalRows struct{} + +var FfiConverterOptionalWalRowsINSTANCE = FfiConverterOptionalWalRows{} + +func (c FfiConverterOptionalWalRows) Lift(rb RustBufferI) *WalRows { + return LiftFromRustBuffer[*WalRows](c, rb) +} + +func (_ FfiConverterOptionalWalRows) Read(reader io.Reader) *WalRows { + if readInt8(reader) == 0 { + return nil + } + temp := FfiConverterWalRowsINSTANCE.Read(reader) + return &temp +} + +func (c FfiConverterOptionalWalRows) Lower(value *WalRows) C.RustBuffer { + return LowerIntoRustBuffer[*WalRows](c, value) +} + +func (c FfiConverterOptionalWalRows) LowerExternal(value *WalRows) ExternalCRustBuffer { + return RustBufferFromC(LowerIntoRustBuffer[*WalRows](c, value)) +} + +func (_ FfiConverterOptionalWalRows) Write(writer io.Writer, value *WalRows) { + if value == nil { + writeInt8(writer, 0) + } else { + writeInt8(writer, 1) + FfiConverterWalRowsINSTANCE.Write(writer, *value) + } +} + +type FfiDestroyerOptionalWalRows struct{} + +func (_ FfiDestroyerOptionalWalRows) Destroy(value *WalRows) { + if value != nil { + FfiDestroyerWalRows{}.Destroy(*value) + } +} + type FfiConverterOptionalCloseReason struct{} var FfiConverterOptionalCloseReasonINSTANCE = FfiConverterOptionalCloseReason{} @@ -14094,53 +14070,6 @@ func (FfiDestroyerSequenceFilterPolicy) Destroy(sequence []*FilterPolicy) { } } -type FfiConverterSequenceWalFile struct{} - -var FfiConverterSequenceWalFileINSTANCE = FfiConverterSequenceWalFile{} - -func (c FfiConverterSequenceWalFile) Lift(rb RustBufferI) []*WalFile { - return LiftFromRustBuffer[[]*WalFile](c, rb) -} - -func (c FfiConverterSequenceWalFile) Read(reader io.Reader) []*WalFile { - length := readInt32(reader) - if length == 0 { - return nil - } - result := make([]*WalFile, 0, length) - for i := int32(0); i < length; i++ { - result = append(result, FfiConverterWalFileINSTANCE.Read(reader)) - } - return result -} - -func (c FfiConverterSequenceWalFile) Lower(value []*WalFile) C.RustBuffer { - return LowerIntoRustBuffer[[]*WalFile](c, value) -} - -func (c FfiConverterSequenceWalFile) LowerExternal(value []*WalFile) ExternalCRustBuffer { - return RustBufferFromC(LowerIntoRustBuffer[[]*WalFile](c, value)) -} - -func (c FfiConverterSequenceWalFile) Write(writer io.Writer, value []*WalFile) { - if len(value) > math.MaxInt32 { - panic("[]*WalFile is too large to fit into Int32") - } - - writeInt32(writer, int32(len(value))) - for _, item := range value { - FfiConverterWalFileINSTANCE.Write(writer, item) - } -} - -type FfiDestroyerSequenceWalFile struct{} - -func (FfiDestroyerSequenceWalFile) Destroy(sequence []*WalFile) { - for _, value := range sequence { - FfiDestroyerWalFile{}.Destroy(value) - } -} - type FfiConverterSequenceCheckpoint struct{} var FfiConverterSequenceCheckpointINSTANCE = FfiConverterSequenceCheckpoint{} @@ -14423,6 +14352,53 @@ func (FfiDestroyerSequenceMetricLabel) Destroy(sequence []MetricLabel) { } } +type FfiConverterSequenceRowEntry struct{} + +var FfiConverterSequenceRowEntryINSTANCE = FfiConverterSequenceRowEntry{} + +func (c FfiConverterSequenceRowEntry) Lift(rb RustBufferI) []RowEntry { + return LiftFromRustBuffer[[]RowEntry](c, rb) +} + +func (c FfiConverterSequenceRowEntry) Read(reader io.Reader) []RowEntry { + length := readInt32(reader) + if length == 0 { + return nil + } + result := make([]RowEntry, 0, length) + for i := int32(0); i < length; i++ { + result = append(result, FfiConverterRowEntryINSTANCE.Read(reader)) + } + return result +} + +func (c FfiConverterSequenceRowEntry) Lower(value []RowEntry) C.RustBuffer { + return LowerIntoRustBuffer[[]RowEntry](c, value) +} + +func (c FfiConverterSequenceRowEntry) LowerExternal(value []RowEntry) ExternalCRustBuffer { + return RustBufferFromC(LowerIntoRustBuffer[[]RowEntry](c, value)) +} + +func (c FfiConverterSequenceRowEntry) Write(writer io.Writer, value []RowEntry) { + if len(value) > math.MaxInt32 { + panic("[]RowEntry is too large to fit into Int32") + } + + writeInt32(writer, int32(len(value))) + for _, item := range value { + FfiConverterRowEntryINSTANCE.Write(writer, item) + } +} + +type FfiDestroyerSequenceRowEntry struct{} + +func (FfiDestroyerSequenceRowEntry) Destroy(sequence []RowEntry) { + for _, value := range sequence { + FfiDestroyerRowEntry{}.Destroy(value) + } +} + type FfiConverterSequenceSegment struct{} var FfiConverterSequenceSegmentINSTANCE = FfiConverterSequenceSegment{} diff --git a/bindings/go/uniffi/slatedb.h b/bindings/go/uniffi/slatedb.h index ed9ac82954..8ad45f4306 100644 --- a/bindings/go/uniffi/slatedb.h +++ b/bindings/go/uniffi/slatedb.h @@ -1642,79 +1642,59 @@ void uniffi_slatedb_uniffi_fn_method_settings_set(uint64_t ptr, RustBuffer key, RustBuffer uniffi_slatedb_uniffi_fn_method_settings_to_json_string(uint64_t ptr, RustCallStatus *out_status ); #endif -#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_CLONE_WALFILE -#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_CLONE_WALFILE -uint64_t uniffi_slatedb_uniffi_fn_clone_walfile(uint64_t handle, RustCallStatus *out_status +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_CLONE_SLATEDBWALITERATOR +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_CLONE_SLATEDBWALITERATOR +uint64_t uniffi_slatedb_uniffi_fn_clone_slatedbwaliterator(uint64_t handle, RustCallStatus *out_status ); #endif -#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_FREE_WALFILE -#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_FREE_WALFILE -void uniffi_slatedb_uniffi_fn_free_walfile(uint64_t handle, RustCallStatus *out_status +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_FREE_SLATEDBWALITERATOR +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_FREE_SLATEDBWALITERATOR +void uniffi_slatedb_uniffi_fn_free_slatedbwaliterator(uint64_t handle, RustCallStatus *out_status ); #endif -#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_WALFILE_ID -#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_WALFILE_ID -uint64_t uniffi_slatedb_uniffi_fn_method_walfile_id(uint64_t ptr, RustCallStatus *out_status +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_SLATEDBWALITERATOR_NEXT +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_SLATEDBWALITERATOR_NEXT +uint64_t uniffi_slatedb_uniffi_fn_method_slatedbwaliterator_next(uint64_t ptr ); #endif -#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_WALFILE_ITERATOR -#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_WALFILE_ITERATOR -uint64_t uniffi_slatedb_uniffi_fn_method_walfile_iterator(uint64_t ptr +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_CLONE_SLATEDBWALREADER +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_CLONE_SLATEDBWALREADER +uint64_t uniffi_slatedb_uniffi_fn_clone_slatedbwalreader(uint64_t handle, RustCallStatus *out_status ); #endif -#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_WALFILE_METADATA -#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_WALFILE_METADATA -uint64_t uniffi_slatedb_uniffi_fn_method_walfile_metadata(uint64_t ptr +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_FREE_SLATEDBWALREADER +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_FREE_SLATEDBWALREADER +void uniffi_slatedb_uniffi_fn_free_slatedbwalreader(uint64_t handle, RustCallStatus *out_status ); #endif -#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_WALFILE_NEXT_FILE -#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_WALFILE_NEXT_FILE -uint64_t uniffi_slatedb_uniffi_fn_method_walfile_next_file(uint64_t ptr, RustCallStatus *out_status +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_CONSTRUCTOR_SLATEDBWALREADER_NEW +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_CONSTRUCTOR_SLATEDBWALREADER_NEW +uint64_t uniffi_slatedb_uniffi_fn_constructor_slatedbwalreader_new(RustBuffer path, uint64_t object_store, RustCallStatus *out_status ); #endif -#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_WALFILE_NEXT_ID -#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_WALFILE_NEXT_ID -uint64_t uniffi_slatedb_uniffi_fn_method_walfile_next_id(uint64_t ptr, RustCallStatus *out_status +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_CONSTRUCTOR_SLATEDBWALREADER_WITH_OPTIONS +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_CONSTRUCTOR_SLATEDBWALREADER_WITH_OPTIONS +uint64_t uniffi_slatedb_uniffi_fn_constructor_slatedbwalreader_with_options(RustBuffer path, uint64_t object_store, RustBuffer options, RustCallStatus *out_status ); #endif -#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_CLONE_WALFILEITERATOR -#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_CLONE_WALFILEITERATOR -uint64_t uniffi_slatedb_uniffi_fn_clone_walfileiterator(uint64_t handle, RustCallStatus *out_status +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_CONSTRUCTOR_SLATEDBWALREADER_WITH_WAL_OBJECT_STORE +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_CONSTRUCTOR_SLATEDBWALREADER_WITH_WAL_OBJECT_STORE +uint64_t uniffi_slatedb_uniffi_fn_constructor_slatedbwalreader_with_wal_object_store(RustBuffer path, uint64_t object_store, uint64_t wal_object_store, RustCallStatus *out_status ); #endif -#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_FREE_WALFILEITERATOR -#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_FREE_WALFILEITERATOR -void uniffi_slatedb_uniffi_fn_free_walfileiterator(uint64_t handle, RustCallStatus *out_status +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_CONSTRUCTOR_SLATEDBWALREADER_WITH_WAL_OBJECT_STORE_AND_OPTIONS +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_CONSTRUCTOR_SLATEDBWALREADER_WITH_WAL_OBJECT_STORE_AND_OPTIONS +uint64_t uniffi_slatedb_uniffi_fn_constructor_slatedbwalreader_with_wal_object_store_and_options(RustBuffer path, uint64_t object_store, uint64_t wal_object_store, RustBuffer options, RustCallStatus *out_status ); #endif -#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_WALFILEITERATOR_NEXT -#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_WALFILEITERATOR_NEXT -uint64_t uniffi_slatedb_uniffi_fn_method_walfileiterator_next(uint64_t ptr +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_SLATEDBWALREADER_ITERATOR +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_SLATEDBWALREADER_ITERATOR +uint64_t uniffi_slatedb_uniffi_fn_method_slatedbwalreader_iterator(uint64_t ptr, uint64_t start_wal_file_id ); #endif -#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_CLONE_WALREADER -#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_CLONE_WALREADER -uint64_t uniffi_slatedb_uniffi_fn_clone_walreader(uint64_t handle, RustCallStatus *out_status -); -#endif -#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_FREE_WALREADER -#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_FREE_WALREADER -void uniffi_slatedb_uniffi_fn_free_walreader(uint64_t handle, RustCallStatus *out_status -); -#endif -#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_CONSTRUCTOR_WALREADER_NEW -#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_CONSTRUCTOR_WALREADER_NEW -uint64_t uniffi_slatedb_uniffi_fn_constructor_walreader_new(RustBuffer path, uint64_t object_store, RustCallStatus *out_status -); -#endif -#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_WALREADER_GET -#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_WALREADER_GET -uint64_t uniffi_slatedb_uniffi_fn_method_walreader_get(uint64_t ptr, uint64_t id, RustCallStatus *out_status -); -#endif -#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_WALREADER_LIST -#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_WALREADER_LIST -uint64_t uniffi_slatedb_uniffi_fn_method_walreader_list(uint64_t ptr, RustBuffer start_id, RustBuffer end_id +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_SLATEDBWALREADER_LAST_WAL_FILE_ID +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_SLATEDBWALREADER_LAST_WAL_FILE_ID +uint64_t uniffi_slatedb_uniffi_fn_method_slatedbwalreader_last_wal_file_id(uint64_t ptr, uint64_t replay_after_wal_id ); #endif #ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_CLONE_WRITEBATCH @@ -2858,51 +2838,21 @@ uint16_t uniffi_slatedb_uniffi_checksum_method_settings_to_json_string(void ); #endif -#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_WALFILE_ID -#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_WALFILE_ID -uint16_t uniffi_slatedb_uniffi_checksum_method_walfile_id(void - -); -#endif -#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_WALFILE_ITERATOR -#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_WALFILE_ITERATOR -uint16_t uniffi_slatedb_uniffi_checksum_method_walfile_iterator(void - -); -#endif -#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_WALFILE_METADATA -#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_WALFILE_METADATA -uint16_t uniffi_slatedb_uniffi_checksum_method_walfile_metadata(void - -); -#endif -#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_WALFILE_NEXT_FILE -#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_WALFILE_NEXT_FILE -uint16_t uniffi_slatedb_uniffi_checksum_method_walfile_next_file(void - -); -#endif -#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_WALFILE_NEXT_ID -#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_WALFILE_NEXT_ID -uint16_t uniffi_slatedb_uniffi_checksum_method_walfile_next_id(void +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_SLATEDBWALITERATOR_NEXT +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_SLATEDBWALITERATOR_NEXT +uint16_t uniffi_slatedb_uniffi_checksum_method_slatedbwaliterator_next(void ); #endif -#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_WALFILEITERATOR_NEXT -#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_WALFILEITERATOR_NEXT -uint16_t uniffi_slatedb_uniffi_checksum_method_walfileiterator_next(void +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_SLATEDBWALREADER_ITERATOR +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_SLATEDBWALREADER_ITERATOR +uint16_t uniffi_slatedb_uniffi_checksum_method_slatedbwalreader_iterator(void ); #endif -#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_WALREADER_GET -#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_WALREADER_GET -uint16_t uniffi_slatedb_uniffi_checksum_method_walreader_get(void - -); -#endif -#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_WALREADER_LIST -#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_WALREADER_LIST -uint16_t uniffi_slatedb_uniffi_checksum_method_walreader_list(void +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_SLATEDBWALREADER_LAST_WAL_FILE_ID +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_SLATEDBWALREADER_LAST_WAL_FILE_ID +uint16_t uniffi_slatedb_uniffi_checksum_method_slatedbwalreader_last_wal_file_id(void ); #endif @@ -3056,9 +3006,27 @@ uint16_t uniffi_slatedb_uniffi_checksum_constructor_settings_load(void ); #endif -#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_CONSTRUCTOR_WALREADER_NEW -#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_CONSTRUCTOR_WALREADER_NEW -uint16_t uniffi_slatedb_uniffi_checksum_constructor_walreader_new(void +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_CONSTRUCTOR_SLATEDBWALREADER_NEW +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_CONSTRUCTOR_SLATEDBWALREADER_NEW +uint16_t uniffi_slatedb_uniffi_checksum_constructor_slatedbwalreader_new(void + +); +#endif +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_CONSTRUCTOR_SLATEDBWALREADER_WITH_OPTIONS +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_CONSTRUCTOR_SLATEDBWALREADER_WITH_OPTIONS +uint16_t uniffi_slatedb_uniffi_checksum_constructor_slatedbwalreader_with_options(void + +); +#endif +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_CONSTRUCTOR_SLATEDBWALREADER_WITH_WAL_OBJECT_STORE +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_CONSTRUCTOR_SLATEDBWALREADER_WITH_WAL_OBJECT_STORE +uint16_t uniffi_slatedb_uniffi_checksum_constructor_slatedbwalreader_with_wal_object_store(void + +); +#endif +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_CONSTRUCTOR_SLATEDBWALREADER_WITH_WAL_OBJECT_STORE_AND_OPTIONS +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_CONSTRUCTOR_SLATEDBWALREADER_WITH_WAL_OBJECT_STORE_AND_OPTIONS +uint16_t uniffi_slatedb_uniffi_checksum_constructor_slatedbwalreader_with_wal_object_store_and_options(void ); #endif diff --git a/bindings/go/uniffi/slatedb_test.go b/bindings/go/uniffi/slatedb_test.go index 79cff3ffb9..3bf89a4d4c 100644 --- a/bindings/go/uniffi/slatedb_test.go +++ b/bindings/go/uniffi/slatedb_test.go @@ -137,12 +137,15 @@ func openTestReader(t *testing.T, store *slatedb.ObjectStore, configure func(*te return handle } -func openTestWalReader(t *testing.T, store *slatedb.ObjectStore) *slatedb.WalReader { +func openTestSlateDbWalReader(t *testing.T, store *slatedb.ObjectStore) *slatedb.SlateDbWalReader { t.Helper() - reader := slatedb.NewWalReader(testDBPath, store) + reader, err := slatedb.NewSlateDbWalReader(testDBPath, store) + if err != nil { + t.Fatalf("NewSlateDbWalReader(): %v", err) + } if reader == nil { - t.Fatal("NewWalReader(): got nil reader") + t.Fatal("NewSlateDbWalReader(): got nil reader") } t.Cleanup(reader.Destroy) @@ -190,20 +193,25 @@ func drainIterator(t *testing.T, iter *slatedb.DbIterator) []slatedb.KeyValue { } } -func drainWalIterator(t *testing.T, iter *slatedb.WalFileIterator) []slatedb.RowEntry { +func readWalBatchesThrough( + t *testing.T, + iter *slatedb.SlateDbWalIterator, + endWalFileID uint64, +) []slatedb.WalRows { t.Helper() - var rows []slatedb.RowEntry - for { - row, err := iter.Next() + var batches []slatedb.WalRows + for len(batches) == 0 || batches[len(batches)-1].LastConsumedWalFileId < endWalFileID { + batch, err := iter.Next() if err != nil { t.Fatalf("wal iterator Next(): %v", err) } - if row == nil { - return rows + if batch == nil { + t.Fatal("live WAL iterator ended unexpectedly") } - rows = append(rows, *row) + batches = append(batches, *batch) } + return batches } func requireRows(t *testing.T, got []slatedb.KeyValue, wantKeys []string, wantValues []string) { @@ -467,6 +475,32 @@ func seedWalFiles(t *testing.T, store *slatedb.ObjectStore) { if err := handle.db.FlushWithOptions(slatedb.FlushOptions{FlushType: slatedb.FlushTypeWal}); err != nil { t.Fatalf("FlushWithOptions(Wal) for merge row: %v", err) } + if err := handle.db.Shutdown(); err != nil { + t.Fatalf("Shutdown() after seeding WAL files: %v", err) + } + handle.open = false +} + +func appendWalValue(t *testing.T, store *slatedb.ObjectStore, key, value string) { + t.Helper() + + handle := openTestDB(t, store, func(t *testing.T, builder *slatedb.DbBuilder) { + t.Helper() + if err := builder.WithMergeOperator(concatMergeOperator{}); err != nil { + t.Fatalf("WithMergeOperator(): %v", err) + } + }) + + if _, err := handle.db.Put([]byte(key), []byte(value)); err != nil { + t.Fatalf("Put(%s): %v", key, err) + } + if err := handle.db.FlushWithOptions(slatedb.FlushOptions{FlushType: slatedb.FlushTypeWal}); err != nil { + t.Fatalf("FlushWithOptions(Wal) for %s: %v", key, err) + } + if err := handle.db.Shutdown(); err != nil { + t.Fatalf("Shutdown() after appending %s: %v", key, err) + } + handle.open = false } func TestDbLifecycleAndStatus(t *testing.T) { @@ -2724,207 +2758,166 @@ func TestAdminDeleteMultipleCheckpoints(t *testing.T) { }) } -func TestWalReaderEmptyStore(t *testing.T) { +func flattenWalRows(batches []slatedb.WalRows) []slatedb.RowEntry { + var rows []slatedb.RowEntry + for _, batch := range batches { + rows = append(rows, batch.Rows...) + } + return rows +} + +func TestWalReaderReportsNoNewFilesAfterCursor(t *testing.T) { store := newMemoryStore(t) - reader := openTestWalReader(t, store) + seedWalFiles(t, store) + reader := openTestSlateDbWalReader(t, store) - files, err := reader.List(nil, nil) + cursor, err := reader.LastWalFileId(0) if err != nil { - t.Fatalf("WalReader.List(nil, nil): %v", err) + t.Fatalf("LastWalFileId(0): %v", err) } - for _, file := range files { - defer file.Destroy() + tail, err := reader.LastWalFileId(cursor) + if err != nil { + t.Fatalf("LastWalFileId(cursor): %v", err) } - - if len(files) != 0 { - t.Fatalf("WalReader.List(nil, nil): got %d files, want 0", len(files)) + if tail != cursor { + t.Fatalf("LastWalFileId(cursor): got %d, want %d", tail, cursor) } } -func TestWalReaderListingAndNavigation(t *testing.T) { +func TestWalReaderStreamsNewWalsThroughOneIterator(t *testing.T) { store := newMemoryStore(t) seedWalFiles(t, store) + reader := openTestSlateDbWalReader(t, store) - reader := openTestWalReader(t, store) - - files, err := reader.List(nil, nil) + firstTail, err := reader.LastWalFileId(0) if err != nil { - t.Fatalf("WalReader.List(nil, nil): %v", err) + t.Fatalf("LastWalFileId(0): %v", err) } - - if len(files) < 3 { - t.Fatalf("WalReader.List(nil, nil): got %d files, want at least 3", len(files)) + iter, err := reader.Iterator(1) + if err != nil { + t.Fatalf("SlateDbWalReader.Iterator(1): %v", err) + } + t.Cleanup(iter.Destroy) + firstBatches := readWalBatchesThrough(t, iter, firstTail) + if len(firstBatches) == 0 { + t.Fatal("initial WAL stream returned no batches") } - ids := make([]uint64, len(files)) - for i, file := range files { - ids[i] = file.Id() - if i > 0 && ids[i] <= ids[i-1] { - t.Fatalf("WalReader.List(nil, nil): ids not ascending: %v", ids) + var previous uint64 + foundEmptyFence := false + for i, batch := range firstBatches { + foundEmptyFence = foundEmptyFence || len(batch.Rows) == 0 + if i > 0 && batch.LastConsumedWalFileId <= previous { + t.Fatalf("WAL cursors did not increase: previous=%d current=%d", previous, batch.LastConsumedWalFileId) } + previous = batch.LastConsumedWalFileId } - - startID := ids[1] - endID := ids[2] - bounded, err := reader.List(&startID, &endID) - if err != nil { - t.Fatalf("WalReader.List(start, end): %v", err) + if !foundEmptyFence { + t.Fatal("initial WAL stream did not return the empty fence WAL batch") } - for _, file := range bounded { - defer file.Destroy() + if previous != firstTail { + t.Fatalf("initial WAL stream ended at %d, want %d", previous, firstTail) } - - if len(bounded) != 1 || bounded[0].Id() != ids[1] { - t.Fatalf("WalReader.List(start, end): got ids [%d], want [%d]", len(bounded), ids[1]) + if tail, err := reader.LastWalFileId(firstTail); err != nil || tail != firstTail { + t.Fatalf("LastWalFileId(firstTail): got tail=%d err=%v, want %d", tail, err, firstTail) } - pastHighID := ids[len(ids)-1] + 1000 - empty, err := reader.List(&pastHighID, nil) + appendWalValue(t, store, "next", "3") + secondTail, err := reader.LastWalFileId(firstTail) if err != nil { - t.Fatalf("WalReader.List(pastHigh, nil): %v", err) + t.Fatalf("LastWalFileId(firstTail) after append: %v", err) } - - if len(empty) != 0 { - t.Fatalf("WalReader.List(pastHigh, nil): got %d files, want 0", len(empty)) + if secondTail <= firstTail { + t.Fatalf("second tail did not advance: first=%d second=%d", firstTail, secondTail) } - - first := reader.Get(ids[0]) - defer first.Destroy() - if first.Id() != ids[0] { - t.Fatalf("WalReader.Get(first): got id %d, want %d", first.Id(), ids[0]) + secondBatches := readWalBatchesThrough(t, iter, secondTail) + if len(secondBatches) == 0 { + t.Fatal("continued WAL stream returned no batches") } - if first.NextId() != ids[1] { - t.Fatalf("WalFile.NextId(): got %d, want %d", first.NextId(), ids[1]) + if got := secondBatches[len(secondBatches)-1].LastConsumedWalFileId; got != secondTail { + t.Fatalf("continued WAL stream ended at %d, want %d", got, secondTail) } - - next := first.NextFile() - defer next.Destroy() - if next.Id() != ids[1] { - t.Fatalf("WalFile.NextFile().Id(): got %d, want %d", next.Id(), ids[1]) + secondRows := flattenWalRows(secondBatches) + if len(secondRows) != 1 || string(secondRows[0].Key) != "next" { + t.Fatalf("continued WAL stream returned rows=%v, want one next row", secondRows) + } + if tail, err := reader.LastWalFileId(secondTail); err != nil || tail != secondTail { + t.Fatalf("LastWalFileId(secondTail): got tail=%d err=%v, want %d", tail, err, secondTail) } } -func TestWalReaderMetadataAndRows(t *testing.T) { +func TestWalReaderDecodesValueTombstoneAndMergeRows(t *testing.T) { store := newMemoryStore(t) seedWalFiles(t, store) + reader := openTestSlateDbWalReader(t, store) - reader := openTestWalReader(t, store) - - files, err := reader.List(nil, nil) + tail, err := reader.LastWalFileId(0) if err != nil { - t.Fatalf("WalReader.List(nil, nil): %v", err) + t.Fatalf("LastWalFileId(0): %v", err) } - for _, file := range files { - defer file.Destroy() - } - - if len(files) < 3 { - t.Fatalf("WalReader.List(nil, nil): got %d files, want at least 3", len(files)) - } - - var allRows []slatedb.RowEntry - nonEmptyFiles := 0 - - for i, file := range files { - metadata, err := file.Metadata() - if err != nil { - t.Fatalf("WalFile.Metadata() for file %d: %v", i, err) - } - if metadata.Id != file.Id() { - t.Fatalf("WalFile.Metadata() for file %d: Id = %d, want %d", i, metadata.Id, file.Id()) - } - if metadata.Metadata.Location == "" { - t.Fatalf("WalFile.Metadata() for file %d: Location is empty", i) - } - - iter, err := file.Iterator() - if err != nil { - t.Fatalf("WalFile.Iterator() for file %d: %v", i, err) - } - t.Cleanup(iter.Destroy) - - rows := drainWalIterator(t, iter) - if metadata.Metadata.Size == 0 { - if len(rows) != 0 { - t.Fatalf("zero-byte WAL file %d returned %d rows, want 0", i, len(rows)) - } - continue - } - nonEmptyFiles++ - - for j, row := range rows { - if row.Seq == 0 { - t.Fatalf("row %d in file %d: Seq = 0", j, i) - } - } - allRows = append(allRows, rows...) + iter, err := reader.Iterator(1) + if err != nil { + t.Fatalf("SlateDbWalReader.Iterator(1): %v", err) } - - if nonEmptyFiles == 0 { - t.Fatal("no non-empty WAL files found") + t.Cleanup(iter.Destroy) + batches := readWalBatchesThrough(t, iter, tail) + if got := batches[len(batches)-1].LastConsumedWalFileId; got != tail { + t.Fatalf("WAL stream ended at %d, want %d", got, tail) } - - if len(allRows) != 4 { - t.Fatalf("unexpected total WAL row count: got %d, want 4", len(allRows)) + rows := flattenWalRows(batches) + if len(rows) != 4 { + t.Fatalf("unexpected total WAL row count: got %d, want 4", len(rows)) } - if allRows[0].Kind != slatedb.RowEntryKindValue || string(allRows[0].Key) != "a" { - t.Fatalf("row 0: got kind=%v key=%q, want value/a", allRows[0].Kind, allRows[0].Key) + if rows[0].Kind != slatedb.RowEntryKindValue || string(rows[0].Key) != "a" { + t.Fatalf("row 0: got kind=%v key=%q, want value/a", rows[0].Kind, rows[0].Key) } - if allRows[0].Value == nil || !bytes.Equal(*allRows[0].Value, []byte("1")) { - t.Fatalf("row 0: got value %v, want %q", allRows[0].Value, "1") + if rows[0].Value == nil || !bytes.Equal(*rows[0].Value, []byte("1")) { + t.Fatalf("row 0: got value %v, want %q", rows[0].Value, "1") } - - if allRows[1].Kind != slatedb.RowEntryKindValue || string(allRows[1].Key) != "b" { - t.Fatalf("row 1: got kind=%v key=%q, want value/b", allRows[1].Kind, allRows[1].Key) + if rows[1].Kind != slatedb.RowEntryKindValue || string(rows[1].Key) != "b" { + t.Fatalf("row 1: got kind=%v key=%q, want value/b", rows[1].Kind, rows[1].Key) } - if allRows[1].Value == nil || !bytes.Equal(*allRows[1].Value, []byte("2")) { - t.Fatalf("row 1: got value %v, want %q", allRows[1].Value, "2") + if rows[1].Value == nil || !bytes.Equal(*rows[1].Value, []byte("2")) { + t.Fatalf("row 1: got value %v, want %q", rows[1].Value, "2") } - - if allRows[2].Kind != slatedb.RowEntryKindTombstone || string(allRows[2].Key) != "a" { - t.Fatalf("row 2: got kind=%v key=%q, want tombstone/a", allRows[2].Kind, allRows[2].Key) + if rows[2].Kind != slatedb.RowEntryKindTombstone || string(rows[2].Key) != "a" { + t.Fatalf("row 2: got kind=%v key=%q, want tombstone/a", rows[2].Kind, rows[2].Key) } - if allRows[2].Value != nil { - t.Fatalf("row 2: got value %q, want nil", *allRows[2].Value) + if rows[2].Value != nil { + t.Fatalf("row 2: got value %q, want nil", *rows[2].Value) } - - if allRows[3].Kind != slatedb.RowEntryKindMerge || string(allRows[3].Key) != "m" { - t.Fatalf("row 3: got kind=%v key=%q, want merge/m", allRows[3].Kind, allRows[3].Key) + if rows[3].Kind != slatedb.RowEntryKindMerge || string(rows[3].Key) != "m" { + t.Fatalf("row 3: got kind=%v key=%q, want merge/m", rows[3].Kind, rows[3].Key) } - if allRows[3].Value == nil || !bytes.Equal(*allRows[3].Value, []byte("x")) { - t.Fatalf("row 3: got value %v, want %q", allRows[3].Value, "x") + if rows[3].Value == nil || !bytes.Equal(*rows[3].Value, []byte("x")) { + t.Fatalf("row 3: got value %v, want %q", rows[3].Value, "x") } } -func TestWalReaderMissingFile(t *testing.T) { +func TestWalReaderCanStartAtTheNextWal(t *testing.T) { store := newMemoryStore(t) seedWalFiles(t, store) + reader := openTestSlateDbWalReader(t, store) - reader := openTestWalReader(t, store) - - files, err := reader.List(nil, nil) + tail, err := reader.LastWalFileId(0) if err != nil { - t.Fatalf("WalReader.List(nil, nil): %v", err) + t.Fatalf("LastWalFileId(0): %v", err) } - for _, file := range files { - defer file.Destroy() - } - - if len(files) == 0 { - t.Fatal("WalReader.List(nil, nil): got 0 files, want at least 1") + iter, err := reader.Iterator(tail + 1) + if err != nil { + t.Fatalf("Iterator(next WAL): %v", err) } + t.Cleanup(iter.Destroy) - missingID := files[len(files)-1].Id() + 1000 - missing := reader.Get(missingID) - defer missing.Destroy() - - if missing.Id() != missingID { - t.Fatalf("WalReader.Get(missing): got id %d, want %d", missing.Id(), missingID) + appendWalValue(t, store, "resumed", "4") + newTail, err := reader.LastWalFileId(tail) + if err != nil { + t.Fatalf("LastWalFileId(tail) after append: %v", err) } - - if _, err := missing.Metadata(); err == nil { - t.Fatal("WalFile.Metadata() for missing file: got nil error, want non-nil error") + rows := flattenWalRows(readWalBatchesThrough(t, iter, newTail)) + if len(rows) != 1 || string(rows[0].Key) != "resumed" { + t.Fatalf("resumed WAL stream returned rows=%v, want one resumed row", rows) } } diff --git a/bindings/java/slatedb-uniffi/src/test/java/io/slatedb/uniffi/SlateDbWalReaderTest.java b/bindings/java/slatedb-uniffi/src/test/java/io/slatedb/uniffi/SlateDbWalReaderTest.java index 979992f0aa..0a65ed75fe 100644 --- a/bindings/java/slatedb-uniffi/src/test/java/io/slatedb/uniffi/SlateDbWalReaderTest.java +++ b/bindings/java/slatedb-uniffi/src/test/java/io/slatedb/uniffi/SlateDbWalReaderTest.java @@ -1,7 +1,7 @@ package io.slatedb.uniffi; import static org.junit.jupiter.api.Assertions.assertEquals; -import static org.junit.jupiter.api.Assertions.assertNotNull; +import static org.junit.jupiter.api.Assertions.assertFalse; import static org.junit.jupiter.api.Assertions.assertTrue; import java.util.ArrayList; @@ -9,137 +9,101 @@ import org.junit.jupiter.api.Test; class SlateDbWalReaderTest { + private static List rowsOf(List batches) { + List rows = new ArrayList<>(); + for (WalRows batch : batches) { + rows.addAll(batch.rows()); + } + return rows; + } + @Test - void walReaderEmptyStore() throws Exception { - try (ObjectStore store = TestSupport.newMemoryStore(); - WalReader reader = TestSupport.openWalReader(store)) { - List files = TestSupport.await(reader.list(null, null)); - try { - assertEquals(0, files.size()); - } finally { - TestSupport.closeAll(files); + void walReaderReportsNoNewFilesAfterCursor() throws Exception { + try (ObjectStore store = TestSupport.newMemoryStore()) { + TestSupport.seedWalFiles(store); + try (SlateDbWalReader reader = TestSupport.openSlateDbWalReader(store)) { + long cursor = TestSupport.await(reader.lastWalFileId(0L)); + assertEquals(cursor, TestSupport.await(reader.lastWalFileId(cursor))); } } } @Test - void walReaderListingAndNavigation() throws Exception { + void walReaderStreamsNewWalsThroughOneIterator() throws Exception { try (ObjectStore store = TestSupport.newMemoryStore()) { TestSupport.seedWalFiles(store); - - try (WalReader reader = TestSupport.openWalReader(store)) { - List files = TestSupport.await(reader.list(null, null)); - try { - assertTrue(files.size() >= 3); - - List ids = new ArrayList<>(); - for (int i = 0; i < files.size(); i++) { - long id = files.get(i).id(); - ids.add(id); - if (i > 0) { - assertTrue(id > ids.get(i - 1)); - } - } - - List bounded = TestSupport.await(reader.list(ids.get(1), ids.get(2))); - try { - assertEquals(1, bounded.size()); - assertEquals(ids.get(1), bounded.get(0).id()); - } finally { - TestSupport.closeAll(bounded); - } - - List empty = TestSupport.await(reader.list(ids.get(ids.size() - 1) + 1000L, null)); - try { - assertEquals(0, empty.size()); - } finally { - TestSupport.closeAll(empty); - } - - try (WalFile first = reader.get(ids.get(0))) { - assertEquals(ids.get(0), first.id()); - assertEquals(ids.get(1), first.nextId()); - - try (WalFile next = first.nextFile()) { - assertEquals(ids.get(1), next.id()); - } + try (SlateDbWalReader reader = TestSupport.openSlateDbWalReader(store)) { + long firstTail = TestSupport.await(reader.lastWalFileId(0L)); + try (SlateDbWalIterator iterator = TestSupport.await(reader.iterator(1L))) { + List firstBatches = + TestSupport.readWalBatchesThrough(iterator, firstTail); + assertFalse(firstBatches.isEmpty()); + assertTrue(firstBatches.stream().anyMatch(batch -> batch.rows().isEmpty())); + long previous = 0L; + for (WalRows batch : firstBatches) { + assertTrue(batch.lastConsumedWalFileId() > previous); + previous = batch.lastConsumedWalFileId(); } - } finally { - TestSupport.closeAll(files); + assertEquals(firstTail, previous); + assertEquals(firstTail, TestSupport.await(reader.lastWalFileId(firstTail))); + + TestSupport.appendWalValue(store, "next", "3"); + long secondTail = TestSupport.await(reader.lastWalFileId(firstTail)); + assertTrue(secondTail > firstTail); + List secondBatches = + TestSupport.readWalBatchesThrough(iterator, secondTail); + assertFalse(secondBatches.isEmpty()); + assertEquals( + secondTail, + secondBatches.get(secondBatches.size() - 1).lastConsumedWalFileId()); + List secondRows = rowsOf(secondBatches); + assertEquals(1, secondRows.size()); + TestSupport.assertWalRow(secondRows.get(0), RowEntryKind.VALUE, "next", "3"); + assertEquals(secondTail, TestSupport.await(reader.lastWalFileId(secondTail))); } } } } @Test - void walReaderMetadataAndRows() throws Exception { + void walReaderDecodesValueTombstoneAndMergeRows() throws Exception { try (ObjectStore store = TestSupport.newMemoryStore()) { TestSupport.seedWalFiles(store); - - try (WalReader reader = TestSupport.openWalReader(store)) { - List files = TestSupport.await(reader.list(null, null)); - try { - assertTrue(files.size() >= 3); - - List allRows = new ArrayList<>(); - int nonEmptyFiles = 0; - for (WalFile file : files) { - IdentifiedObjectMetadata metadata = TestSupport.await(file.metadata()); - assertNotNull(metadata); - assertEquals(file.id(), metadata.id()); - assertFalse(metadata.metadata().location().isEmpty()); - - try (WalFileIterator iterator = TestSupport.await(file.iterator())) { - List rows = TestSupport.drainWalIterator(iterator); - if (metadata.metadata().size() == 0) { - assertTrue(rows.isEmpty()); - continue; - } - - nonEmptyFiles++; - for (RowEntry row : rows) { - assertTrue(row.seq() > 0); - } - allRows.addAll(rows); - } - } - - assertTrue(nonEmptyFiles > 0); - assertEquals(4, allRows.size()); - TestSupport.assertWalRow(allRows.get(0), RowEntryKind.VALUE, "a", "1"); - TestSupport.assertWalRow(allRows.get(1), RowEntryKind.VALUE, "b", "2"); - TestSupport.assertWalRow(allRows.get(2), RowEntryKind.TOMBSTONE, "a", null); - TestSupport.assertWalRow(allRows.get(3), RowEntryKind.MERGE, "m", "x"); - } finally { - TestSupport.closeAll(files); + try (SlateDbWalReader reader = TestSupport.openSlateDbWalReader(store)) { + long tail = TestSupport.await(reader.lastWalFileId(0L)); + List batches; + try (SlateDbWalIterator iterator = TestSupport.await(reader.iterator(1L))) { + batches = TestSupport.readWalBatchesThrough(iterator, tail); + assertEquals(tail, batches.get(batches.size() - 1).lastConsumedWalFileId()); } + + List rows = rowsOf(batches); + assertEquals(4, rows.size()); + assertTrue(rows.stream().allMatch(row -> row.seq() > 0L)); + TestSupport.assertWalRow(rows.get(0), RowEntryKind.VALUE, "a", "1"); + TestSupport.assertWalRow(rows.get(1), RowEntryKind.VALUE, "b", "2"); + TestSupport.assertWalRow(rows.get(2), RowEntryKind.TOMBSTONE, "a", null); + TestSupport.assertWalRow(rows.get(3), RowEntryKind.MERGE, "m", "x"); } } } @Test - void walReaderMissingFile() throws Exception { + void walReaderCanStartAtTheNextWal() throws Exception { try (ObjectStore store = TestSupport.newMemoryStore()) { TestSupport.seedWalFiles(store); - - try (WalReader reader = TestSupport.openWalReader(store)) { - List files = TestSupport.await(reader.list(null, null)); - try { - assertFalse(files.isEmpty()); - long missingId = files.get(files.size() - 1).id() + 1000L; - - try (WalFile missing = reader.get(missingId)) { - assertEquals(missingId, missing.id()); - TestSupport.awaitFailure(Error.class, missing.metadata()); - } - } finally { - TestSupport.closeAll(files); + try (SlateDbWalReader reader = TestSupport.openSlateDbWalReader(store)) { + long tail = TestSupport.await(reader.lastWalFileId(0L)); + try (SlateDbWalIterator iterator = + TestSupport.await(reader.iterator(Math.addExact(tail, 1L)))) { + TestSupport.appendWalValue(store, "resumed", "4"); + long newTail = TestSupport.await(reader.lastWalFileId(tail)); + List rows = rowsOf( + TestSupport.readWalBatchesThrough(iterator, newTail)); + assertEquals(1, rows.size()); + TestSupport.assertWalRow(rows.get(0), RowEntryKind.VALUE, "resumed", "4"); } } } } - - private static void assertFalse(boolean condition) { - org.junit.jupiter.api.Assertions.assertFalse(condition); - } } diff --git a/bindings/java/slatedb-uniffi/src/test/java/io/slatedb/uniffi/TestSupport.java b/bindings/java/slatedb-uniffi/src/test/java/io/slatedb/uniffi/TestSupport.java index 9ffe5af99f..0534dab438 100644 --- a/bindings/java/slatedb-uniffi/src/test/java/io/slatedb/uniffi/TestSupport.java +++ b/bindings/java/slatedb-uniffi/src/test/java/io/slatedb/uniffi/TestSupport.java @@ -182,8 +182,8 @@ static ManagedReader openReader(String path, ObjectStore store, DbReaderBuilderC } } - static WalReader openWalReader(ObjectStore store) throws Exception { - return new WalReader(TEST_DB_PATH, store); + static SlateDbWalReader openSlateDbWalReader(ObjectStore store) throws Exception { + return new SlateDbWalReader(TEST_DB_PATH, store); } static void seedWalFiles(ObjectStore store) throws Exception { @@ -201,6 +201,14 @@ static void seedWalFiles(ObjectStore store) throws Exception { } } + static void appendWalValue(ObjectStore store, String key, String value) throws Exception { + try (ManagedDb handle = openDb(store, builder -> builder.withMergeOperator(new ConcatMergeOperator()))) { + Db db = handle.db(); + await(db.put(bytes(key), bytes(value))); + await(db.flushWithOptions(new FlushOptions(FlushType.WAL))); + } + } + static T await(CompletableFuture future) throws Exception { return future.get(TIMEOUT_SECONDS, TimeUnit.SECONDS); } @@ -273,15 +281,18 @@ static List drainIterator(DbIterator iterator) throws Exception { } } - static List drainWalIterator(WalFileIterator iterator) throws Exception { - List rows = new ArrayList<>(); - while (true) { - RowEntry row = await(iterator.next()); - if (row == null) { - return rows; + static List readWalBatchesThrough( + SlateDbWalIterator iterator, long endWalFileId) throws Exception { + List batches = new ArrayList<>(); + while (batches.isEmpty() + || batches.get(batches.size() - 1).lastConsumedWalFileId() < endWalFileId) { + WalRows batch = await(iterator.next()); + if (batch == null) { + throw new AssertionError("live WAL iterator ended unexpectedly"); } - rows.add(row); + batches.add(batch); } + return batches; } static void closeAll(Iterable closeables) throws Exception { diff --git a/bindings/node/tests/support.mjs b/bindings/node/tests/support.mjs index c90bde437d..1cbd3f56a0 100644 --- a/bindings/node/tests/support.mjs +++ b/bindings/node/tests/support.mjs @@ -10,8 +10,8 @@ import { LogLevel, ObjectStore, RowEntryKind, + SlateDbWalReader, Ttl, - WalReader, } from "../index.js"; export const TEST_DB_PATH = "test-db"; @@ -175,8 +175,8 @@ export async function openReader(store, { path = TEST_DB_PATH, configure, cleanu } } -export function openWalReader(store, { path = TEST_DB_PATH, cleanup } = {}) { - const reader = new WalReader(path, store); +export function openSlateDbWalReader(store, { path = TEST_DB_PATH, cleanup } = {}) { + const reader = new SlateDbWalReader(path, store); return cleanup?.track(reader, { shutdown: false }) ?? reader; } @@ -191,15 +191,14 @@ export async function drainIterator(iterator) { } } -export async function drainWalIterator(iterator) { - const rows = []; - for (;;) { - const row = await iterator.next(); - if (row == null) { - return rows; - } - rows.push(row); +export async function readWalBatchesThrough(iterator, endWalFileId) { + const batches = []; + while (batches.length === 0 || BigInt(batches.at(-1).last_consumed_wal_file_id) < endWalFileId) { + const batch = await iterator.next(); + assert.ok(batch != null, "live WAL iterator ended unexpectedly"); + batches.push(batch); } + return batches; } export function requireRows(rows, wantKeys, wantValues) { @@ -364,6 +363,21 @@ export async function seedWalFiles(store) { } } +export async function appendWalValue(store, key, value) { + const db = await openDb(store, { + configure(builder) { + builder.with_merge_operator(new ConcatMergeOperator()); + }, + }); + + try { + await db.put_with_options(bytes(key), bytes(value), putOptions(), writeOptions()); + await db.flush_with_options({ flush_type: FlushType.Wal }); + } finally { + await shutdownAndDispose(db); + } +} + export function uniquePath(prefix) { return `${prefix}-${randomUUID()}`; } diff --git a/bindings/node/tests/wal-reader.test.mjs b/bindings/node/tests/wal-reader.test.mjs index 716957ddf8..40a51f4f5b 100644 --- a/bindings/node/tests/wal-reader.test.mjs +++ b/bindings/node/tests/wal-reader.test.mjs @@ -1,108 +1,96 @@ import assert from "node:assert/strict"; import test from "node:test"; -import { ErrorData } from "../index.js"; import { RowEntryKind, + appendWalValue, createCleanup, - drainWalIterator, - expectError, newMemoryStore, - openWalReader, + openSlateDbWalReader, + readWalBatchesThrough, requireWalRow, seedWalFiles, } from "./support.mjs"; -test("wal reader empty store listing", async (t) => { +function cursorOf(batch) { + return BigInt(batch.last_consumed_wal_file_id); +} + +function rowsOf(batches) { + return batches.flatMap((batch) => batch.rows); +} + +test("wal reader reports no new files after the cursor", async (t) => { const cleanup = createCleanup(t); const store = cleanup.track(newMemoryStore()); - const reader = openWalReader(store, { cleanup }); + await seedWalFiles(store); + const reader = openSlateDbWalReader(store, { cleanup }); - assert.deepEqual(await reader.list(undefined, undefined), []); + const cursor = BigInt(await reader.last_wal_file_id(0n)); + assert.equal(BigInt(await reader.last_wal_file_id(cursor)), cursor); }); -test("wal reader listing bounds and navigation", async (t) => { +test("wal reader streams new WALs through one iterator", async (t) => { const cleanup = createCleanup(t); const store = cleanup.track(newMemoryStore()); await seedWalFiles(store); - - const reader = openWalReader(store, { cleanup }); - const files = (await reader.list(undefined, undefined)).map((file) => cleanup.track(file)); - assert.ok(files.length >= 3); - - const ids = files.map((file) => BigInt(file.id())); + const reader = openSlateDbWalReader(store, { cleanup }); + + const firstTail = BigInt(await reader.last_wal_file_id(0n)); + const iterator = cleanup.track(await reader.iterator(1n)); + const firstBatches = await readWalBatchesThrough(iterator, firstTail); + assert.ok(firstBatches.length > 0); + assert.ok(firstBatches.some((batch) => batch.rows.length === 0)); + const firstCursors = firstBatches.map(cursorOf); + assert.equal(firstCursors.at(-1), firstTail); + assert.ok(firstCursors.every((cursor, index) => index === 0 || cursor > firstCursors[index - 1])); + assert.equal(BigInt(await reader.last_wal_file_id(firstTail)), firstTail); + + await appendWalValue(store, "next", "3"); + const secondTail = BigInt(await reader.last_wal_file_id(firstTail)); + assert.ok(secondTail > firstTail); + const secondBatches = await readWalBatchesThrough(iterator, secondTail); + assert.ok(secondBatches.length > 0); + assert.equal(cursorOf(secondBatches.at(-1)), secondTail); assert.deepEqual( - ids, - [...ids].sort((left, right) => (left < right ? -1 : left > right ? 1 : 0)), + rowsOf(secondBatches).map((row) => Buffer.from(row.key).toString("utf8")), + ["next"], ); - - const bounded = (await reader.list(ids[1], ids[2])).map((file) => cleanup.track(file)); - assert.deepEqual( - bounded.map((file) => BigInt(file.id())), - [ids[1]], - ); - - assert.deepEqual(await reader.list(ids.at(-1) + 1_000n, undefined), []); - - const first = cleanup.track(reader.get(ids[0])); - assert.equal(BigInt(first.id()), ids[0]); - assert.equal(BigInt(first.next_id()), ids[1]); - - const nextFile = cleanup.track(first.next_file()); - assert.equal(BigInt(nextFile.id()), ids[1]); + assert.equal(BigInt(await reader.last_wal_file_id(secondTail)), secondTail); }); -test("wal reader metadata and row decoding", async (t) => { +test("wal reader decodes value, tombstone, and merge rows", async (t) => { const cleanup = createCleanup(t); const store = cleanup.track(newMemoryStore()); await seedWalFiles(store); - - const reader = openWalReader(store, { cleanup }); - const files = (await reader.list(undefined, undefined)).map((file) => cleanup.track(file)); - assert.ok(files.length >= 3); - - const allRows = []; - let nonEmptyFiles = 0; - for (const walFile of files) { - const metadata = await walFile.metadata(); - assert.equal(BigInt(metadata.id), BigInt(walFile.id())); - assert.notEqual(metadata.metadata.location, ""); - - const iterator = cleanup.track(await walFile.iterator()); - const rows = await drainWalIterator(iterator); - if (BigInt(metadata.metadata.size) === 0n) { - assert.deepEqual(rows, []); - continue; - } - - nonEmptyFiles += 1; - for (const row of rows) { - assert.ok(BigInt(row.seq) > 0n); - } - allRows.push(...rows); - } - - assert.ok(nonEmptyFiles > 0); - assert.equal(allRows.length, 4); - requireWalRow(allRows[0], RowEntryKind.Value, "a", "1"); - requireWalRow(allRows[1], RowEntryKind.Value, "b", "2"); - requireWalRow(allRows[2], RowEntryKind.Tombstone, "a", undefined); - requireWalRow(allRows[3], RowEntryKind.Merge, "m", "x"); + const reader = openSlateDbWalReader(store, { cleanup }); + + const tail = BigInt(await reader.last_wal_file_id(0n)); + const batches = await readWalBatchesThrough(cleanup.track(await reader.iterator(1n)), tail); + assert.equal(cursorOf(batches.at(-1)), tail); + const rows = rowsOf(batches); + + assert.equal(rows.length, 4); + assert.ok(rows.every((row) => BigInt(row.seq) > 0n)); + requireWalRow(rows[0], RowEntryKind.Value, "a", "1"); + requireWalRow(rows[1], RowEntryKind.Value, "b", "2"); + requireWalRow(rows[2], RowEntryKind.Tombstone, "a", undefined); + requireWalRow(rows[3], RowEntryKind.Merge, "m", "x"); }); -test("wal reader missing file metadata failure", async (t) => { +test("wal reader can start at the next WAL", async (t) => { const cleanup = createCleanup(t); const store = cleanup.track(newMemoryStore()); await seedWalFiles(store); + const reader = openSlateDbWalReader(store, { cleanup }); + const tail = BigInt(await reader.last_wal_file_id(0n)); - const reader = openWalReader(store, { cleanup }); - const files = (await reader.list(undefined, undefined)).map((file) => cleanup.track(file)); - assert.ok(files.length > 0); - - const missingId = BigInt(files.at(-1).id()) + 1_000n; - const missing = cleanup.track(reader.get(missingId)); - assert.equal(BigInt(missing.id()), missingId); - - const error = await expectError(() => missing.metadata(), ErrorData); - assert.match(error.message, /not found/); + const iterator = cleanup.track(await reader.iterator(tail + 1n)); + await appendWalValue(store, "resumed", "4"); + const newTail = BigInt(await reader.last_wal_file_id(tail)); + const batches = await readWalBatchesThrough(iterator, newTail); + assert.deepEqual( + rowsOf(batches).map((row) => Buffer.from(row.key).toString("utf8")), + ["resumed"], + ); }); diff --git a/bindings/python/tests/conftest.py b/bindings/python/tests/conftest.py index 5c3a49c038..f71d26683b 100644 --- a/bindings/python/tests/conftest.py +++ b/bindings/python/tests/conftest.py @@ -29,8 +29,9 @@ RowEntry, RowEntryKind, ScanOptions, + SlateDbWalIterator, Ttl, - WalFileIterator, + WalRows, WriteOptions, ) @@ -129,13 +130,15 @@ async def drain_iterator(iterator: DbIterator) -> list[KeyValue]: rows.append(row) -async def drain_wal_iterator(iterator: WalFileIterator) -> list[RowEntry]: - rows: list[RowEntry] = [] - while True: - row = await iterator.next() - if row is None: - return rows - rows.append(row) +async def read_wal_batches_through( + iterator: SlateDbWalIterator, end_wal_file_id: int +) -> list[WalRows]: + batches: list[WalRows] = [] + while not batches or batches[-1].last_consumed_wal_file_id < end_wal_file_id: + batch = await iterator.next() + assert batch is not None, "live WAL iterator ended unexpectedly" + batches.append(batch) + return batches def require_rows(rows: list[KeyValue], want_keys: list[str], want_values: list[str]) -> None: @@ -239,5 +242,14 @@ async def seed_wal_files(store: ObjectStore) -> None: await db.flush_with_options(FlushOptions(flush_type=FlushType.WAL)) +async def append_wal_value(store: ObjectStore, key: bytes, value: bytes) -> None: + async with open_db( + store, + configure=lambda builder: builder.with_merge_operator(ConcatMergeOperator()), + ) as db: + await db.put(key, value) + await db.flush_with_options(FlushOptions(flush_type=FlushType.WAL)) + + def unique_path(prefix: str) -> str: return f"{prefix}-{uuid.uuid4()}" diff --git a/bindings/python/tests/test_wal_reader.py b/bindings/python/tests/test_wal_reader.py index 381d21d09d..1d8ece1118 100644 --- a/bindings/python/tests/test_wal_reader.py +++ b/bindings/python/tests/test_wal_reader.py @@ -3,91 +3,80 @@ import pytest from conftest import ( TEST_DB_PATH, - drain_wal_iterator, + append_wal_value, new_memory_store, + read_wal_batches_through, require_wal_row, seed_wal_files, ) -from slatedb.uniffi import Error, RowEntryKind, WalReader +from slatedb.uniffi import RowEntryKind, SlateDbWalReader @pytest.mark.asyncio -async def test_wal_reader_empty_store_listing() -> None: - reader = WalReader(TEST_DB_PATH, new_memory_store()) - assert await reader.list(None, None) == [] - - -@pytest.mark.asyncio -async def test_wal_reader_listing_bounds_and_navigation() -> None: +async def test_wal_reader_reports_no_new_files_after_cursor() -> None: store = new_memory_store() await seed_wal_files(store) + reader = SlateDbWalReader(TEST_DB_PATH, store) - reader = WalReader(TEST_DB_PATH, store) - files = await reader.list(None, None) - assert len(files) >= 3 - - ids = [wal_file.id() for wal_file in files] - assert ids == sorted(ids) - - bounded = await reader.list(ids[1], ids[2]) - assert [wal_file.id() for wal_file in bounded] == [ids[1]] - - assert await reader.list(ids[-1] + 1_000, None) == [] - - first = reader.get(ids[0]) - assert first.id() == ids[0] - assert first.next_id() == ids[1] - - next_file = first.next_file() - assert next_file.id() == ids[1] + cursor = await reader.last_wal_file_id(0) + assert await reader.last_wal_file_id(cursor) == cursor @pytest.mark.asyncio -async def test_wal_reader_metadata_and_row_decoding() -> None: +async def test_wal_reader_streams_new_wals_through_one_iterator() -> None: store = new_memory_store() await seed_wal_files(store) + reader = SlateDbWalReader(TEST_DB_PATH, store) + + first_tail = await reader.last_wal_file_id(0) + iterator = await reader.iterator(1) + first_batches = await read_wal_batches_through(iterator, first_tail) + assert first_batches + assert any(not batch.rows for batch in first_batches) + first_cursors = [batch.last_consumed_wal_file_id for batch in first_batches] + assert first_cursors[-1] == first_tail + assert first_cursors == sorted(set(first_cursors)) + assert await reader.last_wal_file_id(first_tail) == first_tail + + await append_wal_value(store, b"next", b"3") + second_tail = await reader.last_wal_file_id(first_tail) + assert second_tail > first_tail + second_batches = await read_wal_batches_through(iterator, second_tail) + assert second_batches + assert second_batches[-1].last_consumed_wal_file_id == second_tail + assert [row.key for batch in second_batches for row in batch.rows] == [b"next"] + assert await reader.last_wal_file_id(second_tail) == second_tail - reader = WalReader(TEST_DB_PATH, store) - files = await reader.list(None, None) - assert len(files) >= 3 - all_rows = [] - non_empty_files = 0 - for wal_file in files: - metadata = await wal_file.metadata() - assert metadata.id == wal_file.id() - assert metadata.metadata.location - - rows = await drain_wal_iterator(await wal_file.iterator()) - if metadata.metadata.size == 0: - assert rows == [] - continue +@pytest.mark.asyncio +async def test_wal_reader_decodes_value_tombstone_and_merge_rows() -> None: + store = new_memory_store() + await seed_wal_files(store) + reader = SlateDbWalReader(TEST_DB_PATH, store) - non_empty_files += 1 - assert all(row.seq > 0 for row in rows) - all_rows.extend(rows) + tail = await reader.last_wal_file_id(0) + batches = await read_wal_batches_through(await reader.iterator(1), tail) + assert batches[-1].last_consumed_wal_file_id == tail + rows = [row for batch in batches for row in batch.rows] - assert non_empty_files > 0 - assert len(all_rows) == 4 - require_wal_row(all_rows[0], RowEntryKind.VALUE, "a", "1") - require_wal_row(all_rows[1], RowEntryKind.VALUE, "b", "2") - require_wal_row(all_rows[2], RowEntryKind.TOMBSTONE, "a", None) - require_wal_row(all_rows[3], RowEntryKind.MERGE, "m", "x") + assert len(rows) == 4 + assert all(row.seq > 0 for row in rows) + require_wal_row(rows[0], RowEntryKind.VALUE, "a", "1") + require_wal_row(rows[1], RowEntryKind.VALUE, "b", "2") + require_wal_row(rows[2], RowEntryKind.TOMBSTONE, "a", None) + require_wal_row(rows[3], RowEntryKind.MERGE, "m", "x") @pytest.mark.asyncio -async def test_wal_reader_missing_file_metadata_failure() -> None: +async def test_wal_reader_can_start_at_the_next_wal() -> None: store = new_memory_store() await seed_wal_files(store) - - reader = WalReader(TEST_DB_PATH, store) - files = await reader.list(None, None) - assert files - - missing = reader.get(files[-1].id() + 1_000) - assert missing.id() == files[-1].id() + 1_000 - - with pytest.raises(Error.Data) as exc: - await missing.metadata() - assert "not found" in exc.value.message + reader = SlateDbWalReader(TEST_DB_PATH, store) + tail = await reader.last_wal_file_id(0) + + iterator = await reader.iterator(tail + 1) + await append_wal_value(store, b"resumed", b"4") + new_tail = await reader.last_wal_file_id(tail) + batches = await read_wal_batches_through(iterator, new_tail) + assert [row.key for batch in batches for row in batch.rows] == [b"resumed"] diff --git a/bindings/uniffi/src/error.rs b/bindings/uniffi/src/error.rs index b3ab852216..486aaaba86 100644 --- a/bindings/uniffi/src/error.rs +++ b/bindings/uniffi/src/error.rs @@ -149,3 +149,60 @@ impl From for Error { } } } + +impl From for Error { + fn from(error: slatedb::wal::WalError) -> Self { + let message = error.to_string(); + match error { + slatedb::wal::WalError::Fenced => Error::Closed { + reason: CloseReason::Fenced, + message, + }, + slatedb::wal::WalError::Closed => Error::Closed { + reason: CloseReason::Clean, + message, + }, + slatedb::wal::WalError::Unavailable(_) => Error::Unavailable { message }, + slatedb::wal::WalError::WalTruncated(_) | slatedb::wal::WalError::DataError(_) => { + Error::Data { message } + } + slatedb::wal::WalError::InternalError(_) => Error::Internal { message }, + _ => Error::Internal { message }, + } + } +} + +#[cfg(test)] +mod tests { + use std::sync::Arc; + + use super::*; + + #[test] + fn wal_errors_preserve_binding_categories() { + assert!(matches!( + Error::from(slatedb::wal::WalError::WalTruncated(7)), + Error::Data { .. } + )); + assert!(matches!( + Error::from(slatedb::wal::WalError::Unavailable(Arc::new( + std::io::Error::other("offline") + ))), + Error::Unavailable { .. } + )); + assert!(matches!( + Error::from(slatedb::wal::WalError::Fenced), + Error::Closed { + reason: CloseReason::Fenced, + .. + } + )); + assert!(matches!( + Error::from(slatedb::wal::WalError::Closed), + Error::Closed { + reason: CloseReason::Clean, + .. + } + )); + } +} diff --git a/bindings/uniffi/src/lib.rs b/bindings/uniffi/src/lib.rs index d287b19aeb..59d00be170 100644 --- a/bindings/uniffi/src/lib.rs +++ b/bindings/uniffi/src/lib.rs @@ -53,7 +53,7 @@ pub use types::{ SegmentPrefix, SortedRun, SourceId, SsTableHandle, SsTableId, SsTableInfo, SsTableView, SstType, VersionedCompactions, VersionedManifest, }; -pub use wal_reader::{WalFile, WalFileIterator, WalReader}; +pub use wal_reader::{SlateDbWalIterator, SlateDbWalReader, SlateDbWalReaderOptions, WalRows}; pub use write_batch::WriteBatch; pub use write_handle::WriteHandle; diff --git a/bindings/uniffi/src/wal_reader.rs b/bindings/uniffi/src/wal_reader.rs index 4148c3ffc5..c7ad872f2b 100644 --- a/bindings/uniffi/src/wal_reader.rs +++ b/bindings/uniffi/src/wal_reader.rs @@ -1,65 +1,85 @@ -use std::ops::Bound; use std::sync::Arc; +use slatedb::wal::WalReader as _; use tokio::sync::Mutex; use crate::error::Error; use crate::object_store::ObjectStore; -use crate::types::{IdentifiedObjectMetadata, RowEntry}; - -/// Handle for a single WAL file. -#[derive(uniffi::Object)] -pub struct WalFile { - inner: slatedb::WalFile, +use crate::types::RowEntry; + +/// Options controlling how the native SlateDB WAL reader fetches WAL SSTs. +#[derive(Clone, Debug, uniffi::Record)] +pub struct SlateDbWalReaderOptions { + /// Number of WAL SSTs to preload. + #[uniffi(default = 4)] + pub sst_batch_size: u64, + /// Number of concurrent fetch tasks per WAL SST. + #[uniffi(default = 2)] + pub max_fetch_tasks: u64, + /// Number of bytes to read ahead from each WAL SST. + #[uniffi(default = 1048576)] + pub read_ahead_bytes: u64, } -impl WalFile { - fn new(inner: slatedb::WalFile) -> Self { - Self { inner } +impl Default for SlateDbWalReaderOptions { + fn default() -> Self { + Self { + sst_batch_size: 4, + max_fetch_tasks: 2, + read_ahead_bytes: 1024 * 1024, + } } } -#[uniffi::export] -impl WalFile { - /// Returns the WAL file ID. - pub fn id(&self) -> u64 { - self.inner.id - } +impl TryFrom for slatedb::wal::SlateDbWalReaderOptions { + type Error = Error; - /// Returns the WAL ID immediately after this file. - pub fn next_id(&self) -> u64 { - self.inner.next_id() + fn try_from(options: SlateDbWalReaderOptions) -> Result { + Ok(Self { + sst_batch_size: positive_usize(options.sst_batch_size, "sst_batch_size")?, + max_fetch_tasks: positive_usize(options.max_fetch_tasks, "max_fetch_tasks")?, + read_ahead_bytes: positive_usize(options.read_ahead_bytes, "read_ahead_bytes")?, + }) } +} - /// Returns a handle for the next WAL file ID without checking existence. - pub fn next_file(&self) -> Arc { - Arc::new(WalFile::new(self.inner.next_file())) +fn positive_usize(value: u64, field: &'static str) -> Result { + if value == 0 { + return Err(Error::Invalid { + message: format!("{field} must be greater than zero"), + }); } + usize::try_from(value).map_err(|_| Error::Invalid { + message: format!("{field} is too large for this platform"), + }) } -#[uniffi::export(async_runtime = "tokio")] -impl WalFile { - /// Reads object-store metadata for this WAL file. - pub async fn metadata(&self) -> Result { - let metadata = self.inner.metadata().await?; - Ok(metadata.into()) - } +/// Rows from one fully consumed WAL file. +#[derive(Clone, Debug, PartialEq, Eq, uniffi::Record)] +pub struct WalRows { + /// Rows stored in the WAL file. Empty fence WALs produce an empty vector. + pub rows: Vec, + /// Last WAL file ID fully consumed by this batch. + pub last_consumed_wal_file_id: u64, +} - /// Opens an iterator over raw row entries in this WAL file. - pub async fn iterator(&self) -> Result, Error> { - let iter = self.inner.iterator().await?; - Ok(Arc::new(WalFileIterator::new(iter))) +impl From for WalRows { + fn from(rows: slatedb::wal::WalRows) -> Self { + Self { + rows: rows.rows.into_iter().map(Into::into).collect(), + last_consumed_wal_file_id: rows.last_consumed_wal_file_id, + } } } -/// Iterator over raw row entries stored in a WAL file. +/// Live iterator over SlateDB WAL files starting at a required WAL file ID. #[derive(uniffi::Object)] -pub struct WalFileIterator { - inner: Mutex, +pub struct SlateDbWalIterator { + inner: Mutex>, } -impl WalFileIterator { - fn new(inner: slatedb::WalFileIterator) -> Self { +impl SlateDbWalIterator { + fn new(inner: Box) -> Self { Self { inner: Mutex::new(inner), } @@ -67,52 +87,116 @@ impl WalFileIterator { } #[uniffi::export(async_runtime = "tokio")] -impl WalFileIterator { - /// Returns the next raw row entry from the WAL file. - pub async fn next(&self) -> Result, Error> { - let mut guard = self.inner.lock().await; - Ok(guard.next().await?.map(RowEntry::from)) +impl SlateDbWalIterator { + /// Returns rows from the next fully consumed WAL file. When it reaches the + /// current tail, this call waits for the next WAL file rather than ending. + pub async fn next(&self) -> Result, Error> { + let mut iterator = self.inner.lock().await; + Ok(iterator.next().await?.map(Into::into)) } } -/// Reader for WAL files stored under a database path. +/// CDC reader backed by SlateDB's native live WAL reader. #[derive(uniffi::Object)] -pub struct WalReader { - inner: slatedb::WalReader, +pub struct SlateDbWalReader { + inner: slatedb::wal::SlateDbWalReader, +} + +impl SlateDbWalReader { + fn build( + path: String, + object_store: Arc, + wal_object_store: Option>, + options: SlateDbWalReaderOptions, + ) -> Result, Error> { + let options = options.try_into()?; + let path = slatedb::object_store::path::Path::from(path); + let mut builder = slatedb::wal::SlateDbWalReaderBuilder::new() + .with_object_store(Arc::clone(&object_store.inner)) + .with_path(path) + .with_options(options); + if let Some(wal_object_store) = wal_object_store { + builder = builder.with_wal_object_store(Arc::clone(&wal_object_store.inner)); + } + let inner = builder.build()?; + Ok(Arc::new(Self { inner })) + } } #[uniffi::export] -impl WalReader { - /// Creates a WAL reader for `path` in `object_store`. +impl SlateDbWalReader { + /// Opens a reader when the manifest and WAL use the same object store. #[uniffi::constructor] - pub fn new(path: String, object_store: Arc) -> Arc { - Arc::new(Self { - inner: slatedb::WalReader::new(path, object_store.inner.clone()), - }) + pub fn new(path: String, object_store: Arc) -> Result, Error> { + Self::build(path, object_store, None, SlateDbWalReaderOptions::default()) + } + + /// Opens a reader with explicit fetch options. + #[uniffi::constructor] + pub fn with_options( + path: String, + object_store: Arc, + options: SlateDbWalReaderOptions, + ) -> Result, Error> { + Self::build(path, object_store, None, options) } - /// Returns a handle for the WAL file with the given ID. - pub fn get(&self, id: u64) -> Arc { - Arc::new(WalFile::new(self.inner.get(id))) + /// Opens a reader for a database with a dedicated WAL object store. + #[uniffi::constructor] + pub fn with_wal_object_store( + path: String, + object_store: Arc, + wal_object_store: Arc, + ) -> Result, Error> { + Self::build( + path, + object_store, + Some(wal_object_store), + SlateDbWalReaderOptions::default(), + ) + } + + /// Opens a reader for a dedicated WAL object store with explicit options. + #[uniffi::constructor] + pub fn with_wal_object_store_and_options( + path: String, + object_store: Arc, + wal_object_store: Arc, + options: SlateDbWalReaderOptions, + ) -> Result, Error> { + Self::build(path, object_store, Some(wal_object_store), options) } } #[uniffi::export(async_runtime = "tokio")] -impl WalReader { - /// Lists WAL files in ascending ID order. - /// - /// `start_id` is inclusive and `end_id` is exclusive when provided. - pub async fn list( - &self, - start_id: Option, - end_id: Option, - ) -> Result>, Error> { - let start = start_id.map(Bound::Included).unwrap_or(Bound::Unbounded); - let end = end_id.map(Bound::Excluded).unwrap_or(Bound::Unbounded); - let files = self.inner.list((start, end)).await?; - Ok(files - .into_iter() - .map(|file| Arc::new(WalFile::new(file))) - .collect()) +impl SlateDbWalReader { + /// Returns a snapshot of the current WAL tail after replay_after_wal_id, or + /// the supplied ID when no later WAL file exists. + pub async fn last_wal_file_id(&self, replay_after_wal_id: u64) -> Result { + Ok(self.inner.last_wal_file_id(replay_after_wal_id).await?) + } + + /// Opens a live iterator starting at start_wal_file_id. The iterator waits + /// and polls internally when it reaches the current WAL tail. + pub async fn iterator(&self, start_wal_file_id: u64) -> Result, Error> { + let iterator = self.inner.iterator((start_wal_file_id..).into()).await?; + Ok(Arc::new(SlateDbWalIterator::new(iterator))) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn rejects_zero_reader_options() { + let options = SlateDbWalReaderOptions { + sst_batch_size: 0, + ..SlateDbWalReaderOptions::default() + }; + assert!(matches!( + slatedb::wal::SlateDbWalReaderOptions::try_from(options), + Err(Error::Invalid { .. }) + )); } } diff --git a/examples/src/change_data_capture.rs b/examples/src/change_data_capture.rs index e24566bb4e..b0fc490c01 100644 --- a/examples/src/change_data_capture.rs +++ b/examples/src/change_data_capture.rs @@ -1,14 +1,9 @@ use slatedb::config::{FlushOptions, FlushType}; -use slatedb::object_store::memory::InMemory; -use slatedb::{Db, RowEntry, ValueDeletable, WalFile, WalReader}; +use slatedb::object_store::{memory::InMemory, path::Path}; +use slatedb::wal::{SlateDbWalReaderBuilder, WalReader as _, WalRows}; +use slatedb::{Db, RowEntry, ValueDeletable}; use std::sync::Arc; -#[derive(Debug, Default)] -struct CdcCursor { - wal_id: u64, - last_seq: u64, -} - #[tokio::main] async fn main() -> anyhow::Result<()> { let object_store = Arc::new(InMemory::new()); @@ -18,41 +13,62 @@ async fn main() -> anyhow::Result<()> { db.put(b"user:1", b"alice").await?; db.put(b"user:2", b"bob").await?; db.delete(b"user:2").await?; - db.flush_with_options(FlushOptions { - flush_type: FlushType::Wal, - }) - .await?; + flush_wal(&db).await?; - let wal_reader = WalReader::new(path, object_store); - let mut cursor = CdcCursor::default(); + let wal_reader = SlateDbWalReaderBuilder::new() + .with_object_store(object_store) + .with_path(Path::from(path)) + .build()?; + let mut cursor = 0_u64; + let start_wal_id = cursor + .checked_add(1) + .ok_or_else(|| anyhow::anyhow!("WAL cursor cannot advance"))?; + let mut iterator = wal_reader.iterator((start_wal_id..).into()).await?; - // Use list() once for discovery (or after a long outage). - for wal_file in wal_reader.list(cursor.wal_id..).await? { - emit_wal_file(&wal_file, &mut cursor).await?; + // Drain the writes that already exist. Empty fence WALs still advance the + // cursor, so stop based on emitted rows only after persisting every batch. + let mut emitted_rows = 0; + while emitted_rows < 3 { + let batch = iterator + .next() + .await? + .ok_or_else(|| anyhow::anyhow!("live WAL iterator ended unexpectedly"))?; + emitted_rows += emit_batch(&batch, &mut cursor); } - // Poll by ID to avoid repeated full prefix listings. - let next_file = wal_reader.get(cursor.wal_id + 1); - emit_wal_file(&next_file, &mut cursor).await?; - println!("Persist cursor periodically: {:?}", cursor); + // Keep the same iterator alive. Its next call observes this later WAL; + // callers do not need to discover a new tail or create another iterator. + db.put(b"user:3", b"carol").await?; + flush_wal(&db).await?; + while emitted_rows < 4 { + let batch = iterator + .next() + .await? + .ok_or_else(|| anyhow::anyhow!("live WAL iterator ended unexpectedly"))?; + emitted_rows += emit_batch(&batch, &mut cursor); + } db.close().await?; Ok(()) } -async fn emit_wal_file(wal_file: &WalFile, cursor: &mut CdcCursor) -> anyhow::Result<()> { - let mut iter = wal_file.iterator().await?; - while let Some(row) = iter.next().await? { - if wal_file.id == cursor.wal_id && row.seq <= cursor.last_seq { - continue; - } +async fn flush_wal(db: &Db) -> Result<(), slatedb::Error> { + db.flush_with_options(FlushOptions { + flush_type: FlushType::Wal, + }) + .await +} - emit_row(wal_file.id, &row); - cursor.wal_id = wal_file.id; - cursor.last_seq = row.seq; +fn emit_batch(batch: &WalRows, cursor: &mut u64) -> usize { + for row in &batch.rows { + emit_row(batch.last_consumed_wal_file_id, row); } - Ok(()) + // Persist only after every row in this WAL file has been emitted. This + // also advances across empty fence WALs. + *cursor = batch.last_consumed_wal_file_id; + println!("persist cursor={cursor}"); + batch.rows.len() } fn emit_row(wal_id: u64, row: &RowEntry) { diff --git a/rfcs/0030-pluggable-wal.md b/rfcs/0030-pluggable-wal.md index 1183bd4198..b34cef000d 100644 --- a/rfcs/0030-pluggable-wal.md +++ b/rfcs/0030-pluggable-wal.md @@ -648,39 +648,22 @@ then deletes them. `DbReaderBuilder` initializes `DbReader` with a `WalReader` that it uses to construct iterators for replaying the WAL when loading a checkpoint. -`DbReader` will now also continually stream WAL updates when configured to track the latest -writes. It does this by creating its `WalIterator` with an unbounded end range and blocking on -`next` from its background polling task. If the reader observes a `WalError::WalTruncated` then it +`DbReader` continually discovers WAL updates when configured to track the latest writes. It +resolves the current tail, creates a bounded `WalIterator`, drains it to completion, and then +refreshes the manifest before the next pass. If the reader observes a `WalError::WalTruncated`, it immediately refreshes the manifest. #### CDC -We'll deprecate/remove the current CDC API. Users can use the `WalReader`/`WalIterator` proposed -in this RFC. SlateDB's native `WalReader` will take a buffer size and a poll interval to use when -tailing the current WAL: +We'll deprecate/remove the file-listing CDC API. CDC users can use SlateDB's native +`SlateDbWalReader`, which implements the `WalReader`/`WalIterator` traits proposed in this RFC. -```rust -struct ObjectStoreWalReader { - // ... -} - -impl ObjectStoreWalReader { - pub fn new>( - path: P, - object_store: Arc, - /// The number of WAL Files to prefetch and buffer when streaming the WAL - buffered_files: usize, - /// The interval at which the next WAL file will be polled when streaming the latest updates - poll_interval: Duration - ) { - todo!() - } -} - -impl WalReader for ObjectStoreWalReader { - // ... -} -``` +CDC uses an iterator with an unbounded end range. A consumer keeps the last fully consumed WAL +file ID, creates one iterator over `(cursor + 1)..`, and persists each +`WalRows.last_consumed_wal_file_id`. At the current tail, `WalIterator::next` polls the manifest and +waits for the next WAL file rather than ending, so the caller does not alternate between tail +discovery and iterator creation. Empty fence WALs return an empty batch that still advances the +cursor. After a restart, the consumer creates a new iterator from the persisted cursor plus one. #### Clones diff --git a/slatedb/src/db.rs b/slatedb/src/db.rs index 83b3fda2dc..f1485b1ab6 100644 --- a/slatedb/src/db.rs +++ b/slatedb/src/db.rs @@ -2250,8 +2250,7 @@ mod tests { OnDemandCompactionSchedulerSupplier, StringConcatMergeOperator, }; use crate::types::RowEntry; - use crate::wal::WalError; - use crate::wal_reader::WalReader; + use crate::wal::{SlateDbWalReaderBuilder, WalError, WalReader as _}; use crate::{proptest_util, test_utils, CloseReason, CompactorBuilder, KeyValue}; use async_trait::async_trait; use chrono::{TimeZone, Utc}; @@ -4819,7 +4818,7 @@ mod tests { let mut settings = test_db_options(0, 1024, None); settings.flush_interval = None; - let kv_store = Db::builder(path, main_object_store) + let kv_store = Db::builder(path, Arc::clone(&main_object_store)) .with_settings(settings) .with_wal_object_store(wal_object_store.clone()) .build() @@ -4867,20 +4866,25 @@ mod tests { 0 ); - let wal_reader = WalReader::new(path, wal_object_store); - let wal_files = wal_reader.list(..).await.unwrap(); - assert_eq!(wal_files.len(), 2); // first file is the fencing operation + let wal_reader = SlateDbWalReaderBuilder::new() + .with_object_store(main_object_store) + .with_wal_object_store(wal_object_store) + .with_path(Path::from(path)) + .build() + .unwrap(); + let tail = wal_reader.last_wal_file_id(0).await.unwrap(); + assert_eq!(tail, 2); // first file is the fencing operation let mut rows = Vec::new(); - let mut wal_iter = wal_files[1] // second file contains the actual write - .iterator() + let mut wal_iter = wal_reader + .iterator((1..tail.checked_add(1).unwrap()).into()) .await .expect("expected successful WAL iterator call"); - while let Some(entry) = wal_iter + while let Some(batch) = wal_iter .next() .await .expect("expected successful WAL rows read") { - rows.push(entry); + rows.extend(batch.rows); } assert_eq!(rows.len(), 1); let row = &rows[0]; diff --git a/slatedb/src/db_reader.rs b/slatedb/src/db_reader.rs index b7decbb385..640fa2a603 100644 --- a/slatedb/src/db_reader.rs +++ b/slatedb/src/db_reader.rs @@ -1,3 +1,4 @@ +use crate::wal::slatedb::reader::SlateDbWalReaderOptions; use { crate::{ bytes_range::{ByteRangeBounds, BytesRange}, @@ -182,11 +183,6 @@ impl DbReaderInner { recorder: slatedb_common::metrics::MetricsRecorderHelper, mut manifest: StoredManifest, ) -> Result { - let wal_reader = wal_reader.unwrap_or_else(|| { - Arc::new(crate::wal::slatedb::reader::SlateDbWalReader::new( - Arc::clone(&table_store), - )) - }); let checkpoint = Self::get_or_create_checkpoint(&mut manifest, mode, &options, rand.clone()).await?; let (manifest_id, initial_manifest) = if let Some(checkpoint) = checkpoint.as_ref() { @@ -197,6 +193,22 @@ impl DbReaderInner { } else { (manifest.id(), manifest.manifest().clone()) }; + let status_manager = DbStatusManager::new_with_initial_values( + initial_manifest.core.last_l0_seq, + VersionedManifest::from_manifest(manifest_id, initial_manifest.clone()), + BTreeSet::default(), + ); + let wal_reader = wal_reader.unwrap_or_else(|| { + Arc::new( + crate::wal::slatedb::reader::SlateDbWalReader::new_with_status_manager( + Arc::clone(&table_store), + &status_manager, + Arc::clone(&system_clock), + SlateDbWalReaderOptions::default(), + ), + ) + }); + let initial_state = Arc::new( Self::build_reader_state( checkpoint, @@ -217,13 +229,15 @@ impl DbReaderInner { initial_state.core().last_l0_clock_tick, )); - // initial_state contains the last_committed_seq after WAL replay. in no-wal mode, we can - // simply fallback to last_l0_seq. let initial_durable_seq = initial_state .last_remote_persisted_seq .max(initial_state.core().last_l0_seq); - let status_manager = DbStatusManager::new_with_initial_values( - initial_durable_seq, + status_manager.report_durable_seq( + initial_state + .last_remote_persisted_seq + .max(initial_state.core().last_l0_seq), + ); + status_manager.report_manifest_and_memtable_segments( VersionedManifest::from(initial_state.as_ref()), collect_touched_segments(initial_state.as_ref()), ); @@ -1458,6 +1472,7 @@ impl DbCacheManagerOps for DbReader { #[cfg(test)] mod tests { + use crate::wal::slatedb::reader::SlateDbWalReaderOptions; use { super::{DbReaderMessage, ManifestPoller, ReaderState, WalReplayEnd}, crate::{ @@ -1501,7 +1516,7 @@ mod tests { DbRand, MockSystemClock, }, std::{ - collections::{BTreeMap, VecDeque}, + collections::{BTreeMap, BTreeSet, VecDeque}, sync::{ atomic::{AtomicUsize, Ordering}, Arc, @@ -2415,10 +2430,11 @@ mod tests { let mut core = ManifestCore::new(); core.next_wal_sst_id = 5; + let status_manager = status_manager_for_core(&core); let (last_wal_id, last_committed_seq) = DbReaderInner::replay_wal_into( Arc::clone(&table_store), - &native_wal_reader(&table_store), + &native_wal_reader(&table_store, &status_manager), &DbReaderOptions::default(), &core, &mut into_tables, @@ -2461,10 +2477,11 @@ mod tests { let mut into_tables = VecDeque::new(); let mut core = ManifestCore::new(); core.next_wal_sst_id = 3; + let status_manager = status_manager_for_core(&core); let (last_wal_id, last_committed_seq) = DbReaderInner::replay_wal_into( Arc::clone(&table_store), - &native_wal_reader(&table_store), + &native_wal_reader(&table_store, &status_manager), &DbReaderOptions::default(), &core, &mut into_tables, @@ -2517,10 +2534,11 @@ mod tests { max_memtable_bytes, ..DbReaderOptions::default() }; + let status_manager = status_manager_for_core(&core); let (last_wal_id, last_committed_seq) = DbReaderInner::replay_wal_into( Arc::clone(&table_store), - &native_wal_reader(&table_store), + &native_wal_reader(&table_store, &status_manager), &reader_options, &core, &mut into_tables, @@ -2554,10 +2572,11 @@ mod tests { let mut into_tables = VecDeque::new(); let core = ManifestCore::new(); + let status_manager = status_manager_for_core(&core); let (last_wal_id, last_committed_seq) = DbReaderInner::replay_wal_into( Arc::clone(&table_store), - &native_wal_reader(&table_store), + &native_wal_reader(&table_store, &status_manager), &DbReaderOptions::default(), &core, &mut into_tables, @@ -2586,10 +2605,11 @@ mod tests { let mut into_tables = VecDeque::new(); let core = ManifestCore::new(); + let status_manager = status_manager_for_core(&core); let (last_wal_id, last_committed_seq) = DbReaderInner::replay_wal_into( Arc::clone(&table_store), - &native_wal_reader(&table_store), + &native_wal_reader(&table_store, &status_manager), &DbReaderOptions::default(), &core, &mut into_tables, @@ -2633,10 +2653,11 @@ mod tests { let mut core = ManifestCore::new(); core.last_l0_seq = 8; core.next_wal_sst_id = 5; + let status_manager = status_manager_for_core(&core); let (last_wal_id, last_committed_seq) = DbReaderInner::replay_wal_into( Arc::clone(&table_store), - &native_wal_reader(&table_store), + &native_wal_reader(&table_store, &status_manager), &DbReaderOptions::default(), &core, &mut into_tables, @@ -3079,10 +3100,24 @@ mod tests { } } + fn status_manager_for_core(core: &ManifestCore) -> DbStatusManager { + DbStatusManager::new_with_initial_values( + core.last_l0_seq, + VersionedManifest::from_manifest(1, Manifest::initial(core.clone())), + BTreeSet::default(), + ) + } + fn native_wal_reader( table_store: &Arc, + status_manager: &DbStatusManager, ) -> crate::wal::slatedb::reader::SlateDbWalReader { - crate::wal::slatedb::reader::SlateDbWalReader::new(Arc::clone(table_store)) + crate::wal::slatedb::reader::SlateDbWalReader::new_with_status_manager( + Arc::clone(table_store), + status_manager, + Arc::new(DefaultSystemClock::new()), + SlateDbWalReaderOptions::default(), + ) } fn immutable_memtable( @@ -3259,7 +3294,8 @@ mod tests { oracle.clone(), None, ); - let wal_reader = Arc::new(native_wal_reader(&table_store)); + let status_manager = status_manager_for_core(&stored_manifest.manifest().core); + let wal_reader = Arc::new(native_wal_reader(&table_store, &status_manager)); let inner = DbReaderInner { manifest_store, table_store, @@ -3273,7 +3309,7 @@ mod tests { system_clock: test_provider.system_clock.clone(), oracle, reader, - status_manager: DbStatusManager::new(0), + status_manager, segment_extractor: None, rand: test_provider.rand.clone(), recorder, @@ -3322,6 +3358,7 @@ mod tests { ) -> DbReaderInner { let manifest_store = test_provider.manifest_store(); let table_store = test_provider.table_store(); + let status_manager = status_manager_for_core(current_core); let prior_state = ReaderState { manifest_id: 1, @@ -3347,7 +3384,7 @@ mod tests { oracle.clone(), None, ); - let wal_reader = Arc::new(native_wal_reader(&table_store)); + let wal_reader = Arc::new(native_wal_reader(&table_store, &status_manager)); DbReaderInner { manifest_store, table_store, @@ -3358,7 +3395,7 @@ mod tests { system_clock: test_provider.system_clock.clone(), oracle, reader, - status_manager: DbStatusManager::new(0), + status_manager, segment_extractor: None, rand: test_provider.rand.clone(), recorder, diff --git a/slatedb/src/wal/mod.rs b/slatedb/src/wal/mod.rs index 15546cdf5a..daf7d93bc7 100644 --- a/slatedb/src/wal/mod.rs +++ b/slatedb/src/wal/mod.rs @@ -6,7 +6,7 @@ use futures::future::BoxFuture; use object_store::path::Path; use std::error::Error; use std::fmt::{Display, Formatter}; -use std::ops::{Bound, Range}; +use std::ops::{Bound, Range, RangeFrom}; use std::sync::Arc; use std::time::Duration; @@ -15,6 +15,10 @@ pub(crate) mod slatedb; pub(crate) mod test_utils; pub(crate) mod wal_disabled; +pub use crate::wal::slatedb::reader::{ + SlateDbWalReader, SlateDbWalReaderBuilder, SlateDbWalReaderOptions, +}; + /// A range of WAL File IDs #[derive(Clone, Debug, Eq, PartialEq)] pub struct WalFileRange(pub Bound, pub Bound); @@ -25,6 +29,12 @@ impl From> for WalFileRange { } } +impl From> for WalFileRange { + fn from(range: RangeFrom) -> Self { + WalFileRange(Bound::Included(range.start), Bound::Unbounded) + } +} + impl TryFrom for Range { type Error = (); diff --git a/slatedb/src/wal/slatedb/iterator.rs b/slatedb/src/wal/slatedb/iterator.rs index 7baad7fd2b..aec27d17b3 100644 --- a/slatedb/src/wal/slatedb/iterator.rs +++ b/slatedb/src/wal/slatedb/iterator.rs @@ -1,22 +1,45 @@ use std::collections::VecDeque; -use std::ops::Range; use std::sync::Arc; +use std::time::Duration; use async_trait::async_trait; use log::error; +use slatedb_common::clock::SystemClock; +use tokio::sync::watch; use tokio::task; use tokio::task::JoinHandle; use crate::db_state::SsTableId; +use crate::db_status::DbStatus; use crate::error::SlateDBError; use crate::iter::{EmptyIterator, RowEntryIterator}; -use crate::manifest::SsTableView; +use crate::manifest::store::ManifestStore; +use crate::manifest::{SsTableView, VersionedManifest}; use crate::sst_iter::{SstIterator, SstIteratorOptions}; use crate::tablestore::TableStore; use crate::utils::panic_string; use crate::wal::{WalError, WalIterator as WalIteratorTrait, WalRows}; use crate::RowEntry; +#[async_trait] +pub(crate) trait ManifestReader: Send + Sync + 'static { + async fn manifest(&self) -> Result; +} + +#[async_trait] +impl ManifestReader for watch::Receiver { + async fn manifest(&self) -> Result { + Ok(self.borrow().current_manifest.clone()) + } +} + +#[async_trait] +impl ManifestReader for ManifestStore { + async fn manifest(&self) -> Result { + self.read_latest_manifest().await + } +} + pub(crate) struct SlateDbWalIteratorOptions { /// The number of SSTs to preload while replaying pub(crate) sst_batch_size: usize, @@ -127,14 +150,15 @@ impl CurrentWalFile { /// /// Preloading only opens each WAL SST (footer, index, and any eagerly fetched /// blocks); a file's rows are read out only when it is returned from -/// [`Self::next`], so at most one file's rows are materialized at a time. +/// [`Self::next`], so at most one file's rows are materialized at a time. For an +/// unbounded end, open tasks poll their assigned future WAL IDs until the files +/// appear or the manifest proves that a missing file was truncated. pub(crate) struct SlateDbWalIterator { options: SlateDbWalIteratorOptions, - /// Range of WAL IDs to iterate over - wal_id_range: Range, + end_bound: WalIteratorEndBound, table_store: Arc, next_files: VecDeque>>, - next_wal_id: u64, + next_wal_id: Option, /// The greatest seq returned so far, used to verify that WAL files arrive /// with strictly increasing seq ranges. last_seq: Option, @@ -144,9 +168,41 @@ pub(crate) struct SlateDbWalIterator { current_file: CurrentWalFile, } +#[derive(Clone)] +pub(crate) enum WalIteratorEndBound { + Exclusive(u64), + Unbounded { + manifest_reader: Arc, + poll_interval: Duration, + system_clock: Arc, + }, +} + +impl WalIteratorEndBound { + fn contains(&self, wal_id: u64) -> bool { + match self { + Self::Exclusive(end_wal_id) => wal_id < *end_wal_id, + Self::Unbounded { .. } => true, + } + } +} + +impl std::fmt::Debug for WalIteratorEndBound { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self { + Self::Exclusive(end_wal_id) => f.debug_tuple("Exclusive").field(end_wal_id).finish(), + Self::Unbounded { poll_interval, .. } => f + .debug_struct("Unbounded") + .field("poll_interval", poll_interval) + .finish_non_exhaustive(), + } + } +} + impl SlateDbWalIterator { pub(crate) fn range( - wal_id_range: Range, + from_wal_id: u64, + to_bound: WalIteratorEndBound, options: SlateDbWalIteratorOptions, table_store: Arc, ) -> Result { @@ -154,28 +210,33 @@ impl SlateDbWalIterator { return Err(SlateDBError::InvalidSSTBatchSize(options.sst_batch_size)); } - let next_wal_id = wal_id_range.start; Ok(Self { options, - wal_id_range, + end_bound: to_bound, table_store, next_files: VecDeque::new(), - next_wal_id, + next_wal_id: Some(from_wal_id), last_seq: None, terminal_result: None, current_file: CurrentWalFile::initial(), }) } + fn spawn_opens(&mut self) { + while self.maybe_spawn_open() {} + } + fn maybe_spawn_open(&mut self) -> bool { - if !self.wal_id_range.contains(&self.next_wal_id) + let Some(next_wal_id) = self.next_wal_id else { + return false; + }; + if !self.end_bound.contains(next_wal_id) || self.next_files.len() >= self.options.sst_batch_size { return false; } - let next_wal_id = self.next_wal_id; - self.next_wal_id += 1; + self.next_wal_id = next_wal_id.checked_add(1); async fn try_open_file_iter( wal_id: u64, @@ -218,11 +279,33 @@ impl SlateDbWalIterator { wal_id: u64, sst_iter_options: SstIteratorOptions, table_store: Arc, + end_bound: WalIteratorEndBound, ) -> Result { - match try_open_file_iter(wal_id, sst_iter_options, table_store).await { - Ok(iter) => Ok(iter), - Err(err) if err.has_object_store_not_found() => Err(WalError::WalTruncated(wal_id)), - Err(err) => Err(err.into()), + loop { + match try_open_file_iter(wal_id, sst_iter_options.clone(), Arc::clone(&table_store)) + .await + { + Ok(iter) => return Ok(iter), + Err(err) if err.has_object_store_not_found() => { + let WalIteratorEndBound::Unbounded { + manifest_reader, + poll_interval, + system_clock, + } = &end_bound + else { + return Err(WalError::WalTruncated(wal_id)); + }; + + let manifest = manifest_reader.manifest().await?; + if wal_id < manifest.next_wal_sst_id() { + // This WAL is known to have been written durably in the past, + // so it must have been deleted by GC. + return Err(WalError::WalTruncated(wal_id)); + } + system_clock.sleep(*poll_interval).await; + } + Err(err) => return Err(err.into()), + } } } @@ -230,17 +313,20 @@ impl SlateDbWalIterator { next_wal_id, self.options.sst_iter_options.clone(), Arc::clone(&self.table_store), + self.end_bound.clone(), )); self.next_files.push_back(handle); true } - /// Await the next preloaded WAL file and return an iterator over its rows. - /// Returns `None` when there are no more files to read. + /// Spawns file loading in the background and populates the iterator with the results of loading + /// the next wal file. async fn load_next_file(&mut self) -> Result<(), WalError> { if self.current_file.initialized() { return Ok(()); } + // Populate the pre-load queue first to handle the case where it's initially empty + self.spawn_opens(); // await a mutable ref to the task so that next remains cancel-safe // see https://docs.rs/tokio/latest/tokio/task/struct.JoinHandle.html#cancel-safety let Some(join_handle) = self.next_files.front_mut() else { @@ -249,13 +335,15 @@ impl SlateDbWalIterator { }; let result = join_handle.await; self.next_files.pop_front(); + // Refill the preload queue before returning so the iterator starts loading the next file + self.spawn_opens(); match result { Ok(result) => { self.current_file.advance(result?); Ok(()) } Err(join_err) => { - let task_name = format!("wal_replay[{:?}]", self.wal_id_range); + let task_name = format!("wal_replay[end_bound={:?}]", self.end_bound); let msg = if let Ok(panic_err) = join_err.try_into_panic() { format!( "wal_replay task panicked unexpectedly. [task_name={}, panic={}]", @@ -288,16 +376,14 @@ impl SlateDbWalIterator { impl WalIteratorTrait for SlateDbWalIterator { /// Get the next set of writes from the WAL files in the range. Each returned /// [`WalRows`] holds the rows of one WAL file; a WAL file with no rows - /// yields a batch with empty `rows`. Returns `None` once all WAL files in the - /// range have been read. It is an error if a WAL file in the range is not - /// present. Errors are returned only on calls that return no batch, so rows - /// read from earlier WAL files are never dropped with a later file's error. + /// yields a batch with empty `rows`. A bounded iterator returns `None` once + /// its range has been read; an iterator with an unbounded end polls future + /// WAL files instead. async fn next(&mut self) -> Result, WalError> { if let Some(result) = self.terminal_result.clone() { return result; } - while self.maybe_spawn_open() {} if let Err(err) = self.load_next_file().await { return self.terminate(Err(err)); } @@ -337,30 +423,57 @@ impl WalIteratorTrait for SlateDbWalIterator { } } +impl Drop for SlateDbWalIterator { + fn drop(&mut self) { + for task in self.next_files.drain(..) { + task.abort(); + } + } +} + #[cfg(test)] mod tests { use std::collections::{BTreeMap, BTreeSet}; use std::sync::Arc; + use std::time::Duration; use bytes::Bytes; use object_store::memory::InMemory; use object_store::path::Path; use object_store::ObjectStore; + use slatedb_common::clock::DefaultSystemClock; - use super::{SlateDbWalIterator, SlateDbWalIteratorOptions}; + use super::{SlateDbWalIterator, SlateDbWalIteratorOptions, WalIteratorEndBound}; use crate::block_cache_policy::BlockCachePolicy; use crate::db_state::SsTableId; + use crate::db_status::DbStatusManager; use crate::format::sst::SsTableFormat; + use crate::manifest::{Manifest, ManifestCore, VersionedManifest}; use crate::object_stores::ObjectStores; use crate::tablestore::{TableStore, TableStoreKind}; use crate::types::RowEntry; use crate::wal::{WalError, WalIterator as _}; + fn versioned_manifest(id: u64, next_wal_id: u64) -> VersionedManifest { + let mut core = ManifestCore::new(); + core.next_wal_sst_id = next_wal_id; + VersionedManifest::from_manifest(id, Manifest::initial(core)) + } + + fn status_manager(next_wal_id: u64) -> DbStatusManager { + DbStatusManager::new_with_initial_values( + 0, + versioned_manifest(1, next_wal_id), + BTreeSet::new(), + ) + } + #[tokio::test] async fn should_repeat_terminal_error_for_wal_iterator() { let table_store = test_table_store(); let mut wal_iter = SlateDbWalIterator::range( - 1..2, + 1, + WalIteratorEndBound::Exclusive(2), SlateDbWalIteratorOptions::default(), Arc::clone(&table_store), ) @@ -380,7 +493,8 @@ mod tests { async fn should_repeat_terminal_none_for_wal_iterator() { let table_store = test_table_store(); let mut wal_iter = SlateDbWalIterator::range( - 1..1, + 1, + WalIteratorEndBound::Exclusive(1), SlateDbWalIteratorOptions::default(), Arc::clone(&table_store), ) @@ -390,6 +504,107 @@ mod tests { assert!(wal_iter.next().await.unwrap().is_none()); } + #[tokio::test] + async fn should_honor_an_exclusive_end_bound() { + let table_store = test_table_store(); + table_store.write_wal_fence(1).await.unwrap(); + table_store.write_wal_fence(2).await.unwrap(); + let mut wal_iter = SlateDbWalIterator::range( + 1, + WalIteratorEndBound::Exclusive(2), + SlateDbWalIteratorOptions::default(), + table_store, + ) + .unwrap(); + + let batch = wal_iter.next().await.unwrap().unwrap(); + assert_eq!(batch.last_consumed_wal_file_id, 1); + assert!(batch.rows.is_empty()); + assert!(wal_iter.next().await.unwrap().is_none()); + } + + #[tokio::test(start_paused = true)] + async fn should_poll_future_wals_in_an_unbounded_range() { + let table_store = test_table_store(); + let status_manager = status_manager(1); + let mut wal_iter = SlateDbWalIterator::range( + 1, + WalIteratorEndBound::Unbounded { + manifest_reader: Arc::new(status_manager.subscribe()), + poll_interval: Duration::from_millis(10), + system_clock: Arc::new(DefaultSystemClock::new()), + }, + SlateDbWalIteratorOptions { + sst_batch_size: 2, + ..SlateDbWalIteratorOptions::default() + }, + Arc::clone(&table_store), + ) + .unwrap(); + + assert!( + tokio::time::timeout(Duration::from_millis(30), wal_iter.next()) + .await + .is_err(), + "an unbounded iterator returned before WAL 1 existed" + ); + + table_store.write_wal_fence(1).await.unwrap(); + let first = tokio::time::timeout(Duration::from_millis(100), wal_iter.next()) + .await + .expect("iterator did not observe WAL 1") + .unwrap() + .expect("unbounded iterator returned None"); + assert!(first.rows.is_empty()); + assert_eq!(first.last_consumed_wal_file_id, 1); + + assert!( + tokio::time::timeout(Duration::from_millis(30), wal_iter.next()) + .await + .is_err(), + "an unbounded iterator returned before WAL 2 existed" + ); + + table_store.write_wal_fence(2).await.unwrap(); + let second = tokio::time::timeout(Duration::from_millis(100), wal_iter.next()) + .await + .expect("iterator did not observe WAL 2") + .unwrap() + .expect("unbounded iterator returned None"); + assert!(second.rows.is_empty()); + assert_eq!(second.last_consumed_wal_file_id, 2); + } + + #[tokio::test(start_paused = true)] + async fn should_report_truncation_when_manifest_advances_past_a_missing_wal() { + let table_store = test_table_store(); + let status_manager = status_manager(1); + let mut wal_iter = SlateDbWalIterator::range( + 1, + WalIteratorEndBound::Unbounded { + manifest_reader: Arc::new(status_manager.subscribe()), + poll_interval: Duration::from_millis(10), + system_clock: Arc::new(DefaultSystemClock::new()), + }, + SlateDbWalIteratorOptions::default(), + table_store, + ) + .unwrap(); + + assert!( + tokio::time::timeout(Duration::from_millis(30), wal_iter.next()) + .await + .is_err(), + "the iterator did not poll a future WAL" + ); + + status_manager.report_manifest(versioned_manifest(2, 2)); + let result = tokio::time::timeout(Duration::from_millis(100), wal_iter.next()) + .await + .expect("iterator did not react to the manifest update"); + assert!(matches!(result, Err(WalError::WalTruncated(1)))); + } + #[tokio::test] async fn should_return_atomic_wal_rows_in_increasing_seq_order() { let table_store = test_table_store(); @@ -433,7 +648,8 @@ mod tests { .unwrap(); } let mut wal_iter = SlateDbWalIterator::range( - 1..(wal_file_count + 1), + 1, + WalIteratorEndBound::Exclusive(wal_file_count + 1), SlateDbWalIteratorOptions::default(), Arc::clone(&table_store), ) diff --git a/slatedb/src/wal/slatedb/reader.rs b/slatedb/src/wal/slatedb/reader.rs index 77a89d882f..9e53cf00fe 100644 --- a/slatedb/src/wal/slatedb/reader.rs +++ b/slatedb/src/wal/slatedb/reader.rs @@ -1,20 +1,229 @@ +use std::ops::Bound; use std::sync::Arc; +use std::time::Duration; use async_trait::async_trait; +use log::error; +use object_store::{path::Path, ObjectStore}; +use slatedb_common::clock::{DefaultSystemClock, SystemClock}; +use crate::block_cache_policy::BlockCachePolicy; +use crate::db_status::DbStatusManager; +use crate::error::SlateDBError; +use crate::format::sst::SsTableFormat; use crate::iter::IterationOrder; +use crate::manifest::store::ManifestStore; +use crate::object_stores::ObjectStores; use crate::sst_iter::SstIteratorOptions; -use crate::tablestore::TableStore; -use crate::wal::slatedb::iterator::{SlateDbWalIterator, SlateDbWalIteratorOptions}; +use crate::tablestore::{TableStore, TableStoreKind}; +use crate::wal::slatedb::iterator::{ + ManifestReader, SlateDbWalIterator, SlateDbWalIteratorOptions, WalIteratorEndBound, +}; use crate::wal::{WalError, WalFileRange, WalIterator, WalReader}; -pub(crate) struct SlateDbWalReader { +#[derive(Clone, Debug)] +pub struct SlateDbWalReaderOptions { + /// The number of SSTs to preload while replaying + pub sst_batch_size: usize, + + /// The number of fetch tasks to spawn per sst. Defaults to 2 so there is always a fetch + /// pending while the current data is being consumed. + pub max_fetch_tasks: usize, + + /// The number of bytes to read ahead in each sst. The value is rounded up to the nearest + /// block size when fetching from object storage. The default is 1MB + pub read_ahead_bytes: usize, +} + +impl Default for SlateDbWalReaderOptions { + fn default() -> Self { + Self { + sst_batch_size: 4, + max_fetch_tasks: 2, + read_ahead_bytes: 1024 * 1024, + } + } +} + +impl From for SlateDbWalIteratorOptions { + fn from(options: SlateDbWalReaderOptions) -> Self { + let format = SsTableFormat::default(); + let blocks_to_fetch = options.read_ahead_bytes.div_ceil(format.block_size); + Self { + sst_batch_size: options.sst_batch_size, + sst_iter_options: SstIteratorOptions { + max_fetch_tasks: options.max_fetch_tasks, + blocks_to_fetch, + cache_blocks: false, + cache_metadata: false, + eager_spawn: true, + order: IterationOrder::Ascending, + prefix: None, + filter_context: None, + }, + } + } +} + +/// Builder for a [`SlateDbWalReader`]. +/// +/// Callers must configure both the database path with [`Self::with_path`] and +/// the primary object store with [`Self::with_object_store`]. By default, the +/// primary object store is used for both the manifest and WAL files. Use +/// [`Self::with_wal_object_store`] when WAL files are stored separately. +/// +/// Reader options and the system clock use their defaults when they are not +/// explicitly configured. +pub struct SlateDbWalReaderBuilder { + path: Option, + table_store: Option>, + object_store: Option>, + wal_object_store: Option>, + manifest_reader: Option>, + system_clock: Arc, + options: SlateDbWalReaderOptions, +} + +impl Default for SlateDbWalReaderBuilder { + fn default() -> Self { + Self { + path: None, + table_store: None, + object_store: None, + wal_object_store: None, + manifest_reader: None, + system_clock: Arc::new(DefaultSystemClock::new()), + options: SlateDbWalReaderOptions::default(), + } + } +} + +impl SlateDbWalReaderBuilder { + /// Creates a builder with default reader options and system clock. + pub fn new() -> Self { + Self::default() + } + + /// Sets the root path of the database to read. + pub fn with_path(mut self, path: Path) -> Self { + self.path = Some(path); + self + } + + /// Sets an existing table store for internal construction. + pub(crate) fn with_table_store(mut self, table_store: Arc) -> Self { + self.table_store = Some(table_store); + self + } + + /// Sets the primary object store used to read the manifest and, by + /// default, WAL files. + pub fn with_object_store(mut self, object_store: Arc) -> Self { + self.object_store = Some(object_store); + self + } + + /// Sets a dedicated object store from which WAL files are read. + /// + /// The primary object store configured by [`Self::with_object_store`] + /// remains the source for the database manifest. + pub fn with_wal_object_store(mut self, wal_object_store: Arc) -> Self { + self.wal_object_store = Some(wal_object_store); + self + } + + /// Sets the clock used to wait between polls by live WAL iterators. + pub fn with_system_clock(mut self, system_clock: Arc) -> Self { + self.system_clock = system_clock; + self + } + + /// Sets the options controlling how WAL files are read. + pub fn with_options(mut self, options: SlateDbWalReaderOptions) -> Self { + self.options = options; + self + } + + /// Sets an existing manifest reader for internal construction. + pub(crate) fn with_manifest_reader(mut self, manifest_reader: Arc) -> Self { + self.manifest_reader = Some(manifest_reader); + self + } + + /// Builds a WAL reader from the configured state. + /// + /// # Errors + /// + /// Returns an invalid-configuration error when the database path or + /// primary object store has not been configured. Internal callers may + /// instead provide both a table store and manifest reader. + pub fn build(self) -> Result { + let manifest_reader = match self.manifest_reader { + Some(manifest_reader) => manifest_reader, + None => { + let Some(object_store) = self.object_store.clone() else { + return Err(crate::Error::invalid( + "must specify object store".to_string(), + )); + }; + let Some(path) = self.path.clone() else { + return Err(crate::Error::invalid("must specify db path".to_string())); + }; + Arc::new(ManifestStore::new(&path, object_store)) + } + }; + let table_store = match self.table_store { + Some(table_store) => table_store, + None => { + let Some(object_store) = self.object_store.clone() else { + return Err(crate::Error::invalid( + "must specify object store".to_string(), + )); + }; + let Some(path) = self.path.clone() else { + return Err(crate::Error::invalid("must specify db path".to_string())); + }; + Arc::new(TableStore::new( + ObjectStores::new(object_store, self.wal_object_store.clone()), + SsTableFormat::default(), + path.clone(), + None, + TableStoreKind::Reader, + BlockCachePolicy::default(), + )) + } + }; + Ok(SlateDbWalReader { + table_store, + manifest_reader, + system_clock: self.system_clock, + options: self.options, + }) + } +} + +pub struct SlateDbWalReader { table_store: Arc, + manifest_reader: Arc, + system_clock: Arc, + options: SlateDbWalReaderOptions, } impl SlateDbWalReader { - pub(crate) fn new(table_store: Arc) -> Self { - Self { table_store } + pub(crate) fn new_with_status_manager( + table_store: Arc, + db_status: &DbStatusManager, + system_clock: Arc, + options: SlateDbWalReaderOptions, + ) -> Self { + let manifest_reader: Arc = Arc::new(db_status.subscribe()); + SlateDbWalReaderBuilder::new() + .with_table_store(table_store) + .with_manifest_reader(manifest_reader) + .with_system_clock(system_clock) + .with_options(options) + .build() + .expect("table store and manifest reader initialize a WAL reader") } } @@ -24,57 +233,433 @@ impl WalReader for SlateDbWalReader { &self, wal_file_id_range: WalFileRange, ) -> Result, WalError> { - let wal_id_range = wal_file_id_range.try_into().map_err(|()| { - WalError::InternalError(Arc::new(std::io::Error::new( - std::io::ErrorKind::InvalidInput, - "native WAL reader requires an included start and excluded end", - ))) - })?; - let iterator = SlateDbWalIterator::range( - wal_id_range, - SlateDbWalIteratorOptions { - sst_batch_size: 4, - sst_iter_options: SstIteratorOptions { - max_fetch_tasks: 1, - blocks_to_fetch: 256, - cache_blocks: true, - cache_metadata: false, - eager_spawn: true, - order: IterationOrder::Ascending, - prefix: None, - filter_context: None, - }, + let from_wal_id = match wal_file_id_range.0 { + Bound::Included(wal_id) => wal_id, + Bound::Excluded(wal_id) => wal_id.checked_add(1).ok_or_else(|| { + error!( + "WAL iterator start bound overflowed. [range={:?}]", + wal_file_id_range + ); + SlateDBError::InvalidDBState + })?, + Bound::Unbounded => { + error!( + "WAL iterator range must have a bounded start. [range={:?}]", + wal_file_id_range + ); + return Err(SlateDBError::InvalidDBState.into()); + } + }; + let end_bound = match wal_file_id_range.1 { + Bound::Included(wal_id) => { + WalIteratorEndBound::Exclusive(wal_id.checked_add(1).ok_or_else(|| { + error!( + "WAL iterator end bound overflowed. [range={:?}]", + wal_file_id_range + ); + SlateDBError::InvalidDBState + })?) + } + Bound::Excluded(wal_id) => WalIteratorEndBound::Exclusive(wal_id), + Bound::Unbounded => WalIteratorEndBound::Unbounded { + manifest_reader: Arc::clone(&self.manifest_reader), + poll_interval: Duration::from_secs(1), + system_clock: Arc::clone(&self.system_clock), }, + }; + let iterator = SlateDbWalIterator::range( + from_wal_id, + end_bound, + self.options.clone().into(), Arc::clone(&self.table_store), )?; Ok(Box::new(iterator)) } async fn last_wal_file_id(&self, replay_after_wal_id: u64) -> Result { - Ok(self + let last = self .table_store .last_seen_wal_id(replay_after_wal_id) - .await?) + .await?; + let manifest = self.manifest_reader.manifest().await?; + if last < manifest.core().replay_after_wal_id { + return Err(WalError::WalTruncated(last)); + } + Ok(last) } } #[cfg(test)] mod tests { + use std::ops::Bound; + use std::time::Duration; + use super::*; - use crate::block_cache_policy::BlockCachePolicy; use crate::config::{FlushOptions, FlushType}; - use crate::format::sst::SsTableFormat; - use crate::object_stores::ObjectStores; - use crate::tablestore::TableStoreKind; + use crate::db_state::SsTableId; + use crate::manifest::store::StoredManifest; + use crate::manifest::ManifestCore; + use crate::paths::PathResolver; + use crate::test_utils::StringConcatMergeOperator; use crate::types::ValueDeletable; + use crate::wal::WalRows; use crate::Db; use object_store::memory::InMemory; - use object_store::{path::Path, ObjectStore}; + use object_store::ObjectStoreExt; + fn end_after(wal_id: u64) -> u64 { + wal_id.checked_add(1).expect("test WAL ID overflow") + } + + fn assert_invalid_build(builder: SlateDbWalReaderBuilder, expected_message: &str) { + let error = match builder.build() { + Ok(_) => panic!("expected WAL reader builder to reject incomplete state"), + Err(error) => error, + }; + assert_eq!(error.kind(), crate::ErrorKind::Invalid); + assert!( + error.to_string().contains(expected_message), + "expected error containing {expected_message:?}, got {error}" + ); + } + + #[test] + fn builder_rejects_missing_object_store() { + assert_invalid_build( + SlateDbWalReaderBuilder::new().with_path(Path::from("/missing-object-store")), + "must specify object store", + ); + } + + #[test] + fn builder_rejects_missing_db_path() { + assert_invalid_build( + SlateDbWalReaderBuilder::new().with_object_store(Arc::new(InMemory::new())), + "must specify db path", + ); + } + + #[test] + fn builder_rejects_table_store_without_manifest_reader() { + let object_store: Arc = Arc::new(InMemory::new()); + let table_store = Arc::new(TableStore::new( + ObjectStores::new(object_store, None), + SsTableFormat::default(), + Path::from("/table-store-without-manifest-reader"), + None, + TableStoreKind::Reader, + BlockCachePolicy::default(), + )); + + assert_invalid_build( + SlateDbWalReaderBuilder::new().with_table_store(table_store), + "must specify object store", + ); + } + + #[test] + fn builder_rejects_manifest_reader_without_table_store() { + let object_store: Arc = Arc::new(InMemory::new()); + let path = Path::from("/manifest-reader-without-table-store"); + let manifest_reader: Arc = + Arc::new(ManifestStore::new(&path, object_store)); + + assert_invalid_build( + SlateDbWalReaderBuilder::new().with_manifest_reader(manifest_reader), + "must specify object store", + ); + } + + #[test] + fn builder_accepts_table_store_and_manifest_reader() { + let object_store: Arc = Arc::new(InMemory::new()); + let path = Path::from("/table-store-and-manifest-reader"); + let table_store = Arc::new(TableStore::new( + ObjectStores::new(Arc::clone(&object_store), None), + SsTableFormat::default(), + path.clone(), + None, + TableStoreKind::Reader, + BlockCachePolicy::default(), + )); + let manifest_reader: Arc = + Arc::new(ManifestStore::new(&path, object_store)); + + assert!(SlateDbWalReaderBuilder::new() + .with_table_store(table_store) + .with_manifest_reader(manifest_reader) + .build() + .is_ok()); + } + + async fn collect_batches( + wal_reader: &SlateDbWalReader, + start_wal_id: u64, + end_wal_id_exclusive: u64, + ) -> Result, WalError> { + let mut iterator = wal_reader + .iterator((start_wal_id..end_wal_id_exclusive).into()) + .await?; + let mut batches = Vec::new(); + while let Some(batch) = iterator.next().await? { + batches.push(batch); + } + Ok(batches) + } + + #[tokio::test] + async fn bounded_polling_discovers_new_wals_and_advances_the_cursor() { + let object_store: Arc = Arc::new(InMemory::new()); + let path = Path::from("/bounded_polling_discovers_new_wals"); + let db = Db::open(path.clone(), Arc::clone(&object_store)) + .await + .unwrap(); + db.put(b"first", b"value-1").await.unwrap(); + db.flush_with_options(FlushOptions { + flush_type: FlushType::Wal, + }) + .await + .unwrap(); + + let wal_reader = SlateDbWalReaderBuilder::new() + .with_object_store(Arc::clone(&object_store)) + .with_path(path) + .build() + .unwrap(); + let first_tail = wal_reader.last_wal_file_id(0).await.unwrap(); + let first_batches = collect_batches(&wal_reader, 1, end_after(first_tail)) + .await + .unwrap(); + let first_cursors: Vec<_> = first_batches + .iter() + .map(|batch| batch.last_consumed_wal_file_id) + .collect(); + assert_eq!(first_cursors, (1..=first_tail).collect::>()); + let first_rows: Vec<_> = first_batches + .iter() + .flat_map(|batch| batch.rows.iter()) + .collect(); + assert_eq!(first_rows.len(), 1); + assert_eq!(first_rows[0].key.as_ref(), b"first"); + + let mut cursor = first_batches.last().unwrap().last_consumed_wal_file_id; + assert_eq!(cursor, first_tail); + assert_eq!(wal_reader.last_wal_file_id(cursor).await.unwrap(), cursor); + + db.put(b"second", b"value-2").await.unwrap(); + db.flush_with_options(FlushOptions { + flush_type: FlushType::Wal, + }) + .await + .unwrap(); + + let second_tail = wal_reader.last_wal_file_id(cursor).await.unwrap(); + assert!(second_tail > cursor); + let second_batches = collect_batches( + &wal_reader, + cursor.checked_add(1).unwrap(), + end_after(second_tail), + ) + .await + .unwrap(); + let second_rows: Vec<_> = second_batches + .iter() + .flat_map(|batch| batch.rows.iter()) + .collect(); + assert_eq!(second_rows.len(), 1); + assert_eq!(second_rows[0].key.as_ref(), b"second"); + cursor = second_batches.last().unwrap().last_consumed_wal_file_id; + assert_eq!(cursor, second_tail); + assert_eq!(wal_reader.last_wal_file_id(cursor).await.unwrap(), cursor); + } + + #[tokio::test] + async fn bounded_iteration_preserves_value_tombstone_merge_and_sequence_order() { + let object_store: Arc = Arc::new(InMemory::new()); + let path = Path::from("/bounded_iteration_preserves_row_kinds"); + let db = Db::builder(path.clone(), Arc::clone(&object_store)) + .with_merge_operator(Arc::new(StringConcatMergeOperator)) + .build() + .await + .unwrap(); + + db.put(b"a", b"1").await.unwrap(); + db.put(b"b", b"2").await.unwrap(); + db.flush_with_options(FlushOptions { + flush_type: FlushType::Wal, + }) + .await + .unwrap(); + db.delete(b"a").await.unwrap(); + db.flush_with_options(FlushOptions { + flush_type: FlushType::Wal, + }) + .await + .unwrap(); + db.merge(b"m", b"x").await.unwrap(); + db.flush_with_options(FlushOptions { + flush_type: FlushType::Wal, + }) + .await + .unwrap(); + + let wal_reader = SlateDbWalReaderBuilder::new() + .with_object_store(object_store) + .with_path(path) + .build() + .unwrap(); + let tail = wal_reader.last_wal_file_id(0).await.unwrap(); + let batches = collect_batches(&wal_reader, 1, end_after(tail)) + .await + .unwrap(); + assert_eq!(batches.last().unwrap().last_consumed_wal_file_id, tail); + let rows: Vec<_> = batches.into_iter().flat_map(|batch| batch.rows).collect(); + assert_eq!(rows.len(), 4); + assert!(rows.windows(2).all(|pair| pair[0].seq < pair[1].seq)); + assert_eq!(rows[0].key.as_ref(), b"a"); + assert!(matches!( + &rows[0].value, + ValueDeletable::Value(value) if value.as_ref() == b"1" + )); + assert_eq!(rows[1].key.as_ref(), b"b"); + assert!(matches!( + &rows[1].value, + ValueDeletable::Value(value) if value.as_ref() == b"2" + )); + assert_eq!(rows[2].key.as_ref(), b"a"); + assert!(matches!(rows[2].value, ValueDeletable::Tombstone)); + assert_eq!(rows[3].key.as_ref(), b"m"); + assert!(matches!( + &rows[3].value, + ValueDeletable::Merge(value) if value.as_ref() == b"x" + )); + } + + #[tokio::test] + async fn empty_fence_wal_advances_the_cursor() { + let object_store: Arc = Arc::new(InMemory::new()); + let path = Path::from("/empty_fence_wal_advances_the_cursor"); + let _db = Db::open(path.clone(), Arc::clone(&object_store)) + .await + .unwrap(); + + let wal_reader = SlateDbWalReaderBuilder::new() + .with_object_store(object_store) + .with_path(path) + .build() + .unwrap(); + let tail = wal_reader.last_wal_file_id(0).await.unwrap(); + let batches = collect_batches(&wal_reader, 1, end_after(tail)) + .await + .unwrap(); + + assert_eq!(tail, 1); + assert_eq!(batches.len(), 1); + assert!(batches[0].rows.is_empty()); + assert_eq!(batches[0].last_consumed_wal_file_id, 1); + } + + #[tokio::test] + async fn should_accept_an_unbounded_end_range() { + let object_store: Arc = Arc::new(InMemory::new()); + let path = Path::from("/reader_accepts_an_unbounded_end_range"); + let _db = Db::open(path.clone(), Arc::clone(&object_store)) + .await + .unwrap(); + let wal_reader = SlateDbWalReaderBuilder::new() + .with_object_store(object_store) + .with_path(path) + .build() + .unwrap(); + let mut iterator = wal_reader.iterator((1..).into()).await.unwrap(); + + let first = tokio::time::timeout(Duration::from_secs(1), iterator.next()) + .await + .expect("unbounded iterator did not observe the fence WAL") + .unwrap() + .expect("unbounded iterator returned None"); + assert!(first.rows.is_empty()); + assert_eq!(first.last_consumed_wal_file_id, 1); + } + + #[tokio::test] + async fn should_reject_an_unbounded_start_range() { + let object_store: Arc = Arc::new(InMemory::new()); + let wal_reader = SlateDbWalReaderBuilder::new() + .with_object_store(object_store) + .with_path(Path::from("/reader_rejects_an_unbounded_start_range")) + .build() + .unwrap(); + + let result = wal_reader + .iterator(WalFileRange(Bound::Unbounded, Bound::Unbounded)) + .await; + assert!(result.is_err()); + } + + #[tokio::test] + async fn should_normalize_excluded_start_and_included_end_bounds() { + let object_store: Arc = Arc::new(InMemory::new()); + let wal_reader = SlateDbWalReaderBuilder::new() + .with_object_store(object_store) + .with_path(Path::from("/reader_normalizes_range_bounds")) + .build() + .unwrap(); + wal_reader.table_store.write_wal_fence(1).await.unwrap(); + wal_reader.table_store.write_wal_fence(2).await.unwrap(); + + let mut iterator = wal_reader + .iterator(WalFileRange(Bound::Excluded(1), Bound::Included(2))) + .await + .unwrap(); + let batch = iterator.next().await.unwrap().unwrap(); + assert_eq!(batch.last_consumed_wal_file_id, 2); + assert!(batch.rows.is_empty()); + assert!(iterator.next().await.unwrap().is_none()); + } #[tokio::test] - async fn test_native_wal_reader_trait() { + async fn reads_manifest_and_wals_from_separate_object_stores() { let object_store: Arc = Arc::new(InMemory::new()); - let path = Path::from("/test_native_wal_reader_trait"); + let wal_object_store: Arc = Arc::new(InMemory::new()); + let path = Path::from("/reader_with_dedicated_wal_store"); + let db = Db::builder(path.clone(), Arc::clone(&object_store)) + .with_wal_object_store(Arc::clone(&wal_object_store)) + .build() + .await + .unwrap(); + db.put(b"dedicated", b"wal-store").await.unwrap(); + db.flush_with_options(FlushOptions { + flush_type: FlushType::Wal, + }) + .await + .unwrap(); + + let wal_reader = SlateDbWalReaderBuilder::new() + .with_object_store(object_store) + .with_wal_object_store(wal_object_store) + .with_path(path) + .build() + .unwrap(); + let tail = wal_reader.last_wal_file_id(0).await.unwrap(); + let rows: Vec<_> = collect_batches(&wal_reader, 1, end_after(tail)) + .await + .unwrap() + .into_iter() + .flat_map(|batch| batch.rows) + .collect(); + + assert_eq!(rows.len(), 1); + assert_eq!(rows[0].key.as_ref(), b"dedicated"); + assert!(matches!( + &rows[0].value, + ValueDeletable::Value(value) if value.as_ref() == b"wal-store" + )); + } + + #[tokio::test] + async fn missing_wal_in_a_bounded_range_returns_truncation() { + let object_store: Arc = Arc::new(InMemory::new()); + let path = Path::from("/missing_wal_in_a_bounded_range"); let db = Db::open(path.clone(), Arc::clone(&object_store)) .await .unwrap(); @@ -85,30 +670,62 @@ mod tests { .await .unwrap(); + let wal_reader = SlateDbWalReaderBuilder::new() + .with_object_store(Arc::clone(&object_store)) + .with_path(path.clone()) + .build() + .unwrap(); + let tail = wal_reader.last_wal_file_id(0).await.unwrap(); + let wal_path = PathResolver::from_root(path).sst_path(&SsTableId::Wal(tail)); + object_store.delete(&wal_path).await.unwrap(); + + let mut iterator = wal_reader + .iterator((tail..end_after(tail)).into()) + .await + .unwrap(); + assert!(matches!( + iterator.next().await, + Err(WalError::WalTruncated(wal_id)) if wal_id == tail + )); + } + + #[tokio::test] + async fn last_wal_file_id_errors_when_last_id_precedes_manifest_gc_cutoff() { + let object_store: Arc = Arc::new(InMemory::new()); + let path = Path::from("/last_wal_file_id_before_gc_cutoff"); let table_store = Arc::new(TableStore::new( - ObjectStores::new(object_store, None), + ObjectStores::new(Arc::clone(&object_store), None), SsTableFormat::default(), - path, + path.clone(), None, TableStoreKind::Reader, BlockCachePolicy::default(), )); - let wal_reader = SlateDbWalReader::new(table_store); - let last_wal_id = wal_reader.last_wal_file_id(0).await.unwrap(); - let mut iterator = wal_reader - .iterator((1..last_wal_id + 1).into()) + table_store + .table_writer(SsTableId::Wal(1)) + .close() .await .unwrap(); - let mut rows = Vec::new(); - while let Some(wal_rows) = iterator.next().await.unwrap() { - rows.extend(wal_rows.rows); - } - assert_eq!(rows.len(), 1); - assert_eq!(rows[0].key.as_ref(), b"key"); + let mut core = ManifestCore::new(); + core.next_wal_sst_id = 3; + core.replay_after_wal_id = 2; + StoredManifest::create_new_db( + Arc::new(ManifestStore::new(&path, Arc::clone(&object_store))), + core, + Arc::new(DefaultSystemClock::new()), + ) + .await + .unwrap(); + + let wal_reader = SlateDbWalReaderBuilder::new() + .with_object_store(object_store) + .with_path(path) + .build() + .unwrap(); assert!(matches!( - &rows[0].value, - ValueDeletable::Value(value) if value.as_ref() == b"value" + wal_reader.last_wal_file_id(0).await, + Err(WalError::WalTruncated(1)) )); } } diff --git a/slatedb/src/wal/slatedb/writer_init.rs b/slatedb/src/wal/slatedb/writer_init.rs index ec1a6be609..59404802ba 100644 --- a/slatedb/src/wal/slatedb/writer_init.rs +++ b/slatedb/src/wal/slatedb/writer_init.rs @@ -5,7 +5,9 @@ use crate::manifest::Manifest; use crate::sst_iter::SstIteratorOptions; use crate::tablestore::TableStore; use crate::utils::WatchableOnceCellReader; -use crate::wal::slatedb::iterator::{SlateDbWalIterator, SlateDbWalIteratorOptions}; +use crate::wal::slatedb::iterator::{ + SlateDbWalIterator, SlateDbWalIteratorOptions, WalIteratorEndBound, +}; use crate::wal::slatedb::writer::SlateDbWalWriter; use crate::wal::{WalError, WriterInitResult, WriterManifest}; use crate::{wal, Settings}; @@ -117,11 +119,12 @@ impl wal::WriterInit for SlateDbWalWriterInit { let replay_after_wal_id = manifest.core().replay_after_wal_id; assert!(empty_wal_id > replay_after_wal_id); let replay_iterator = SlateDbWalIterator::range( - replay_after_wal_id + 1..empty_wal_id + 1, + replay_after_wal_id + 1, + WalIteratorEndBound::Exclusive(empty_wal_id + 1), SlateDbWalIteratorOptions { sst_batch_size: 4, sst_iter_options: SstIteratorOptions { - max_fetch_tasks: 1, + max_fetch_tasks: 2, blocks_to_fetch: 256, cache_blocks: false, cache_metadata: false, diff --git a/slatedb/src/wal_replay.rs b/slatedb/src/wal_replay.rs index afd98618de..0344e834fd 100644 --- a/slatedb/src/wal_replay.rs +++ b/slatedb/src/wal_replay.rs @@ -3,7 +3,9 @@ use crate::manifest::ManifestCore; use crate::mem_table::WritableKVTable; use crate::tablestore::TableStore; #[cfg(test)] -use crate::wal::slatedb::iterator::{SlateDbWalIterator, SlateDbWalIteratorOptions}; +use crate::wal::slatedb::iterator::{ + SlateDbWalIterator, SlateDbWalIteratorOptions, WalIteratorEndBound, +}; use crate::wal::WalIterator as WalIteratorTrait; #[cfg(test)] use std::ops::Range; @@ -60,8 +62,12 @@ impl WalReplayIterator { replay_options: WalReplayOptions, table_store: Arc, ) -> Result { - let wal_iter = - SlateDbWalIterator::range(wal_id_range, iterator_options, Arc::clone(&table_store))?; + let wal_iter = SlateDbWalIterator::range( + wal_id_range.start, + WalIteratorEndBound::Exclusive(wal_id_range.end), + iterator_options, + Arc::clone(&table_store), + )?; Self::for_wal_iterator(Box::new(wal_iter), db_state, replay_options, table_store) } diff --git a/website/src/content/docs/docs/design/change-data-capture.mdx b/website/src/content/docs/docs/design/change-data-capture.mdx index a569ca3589..c534649b4b 100644 --- a/website/src/content/docs/docs/design/change-data-capture.mdx +++ b/website/src/content/docs/docs/design/change-data-capture.mdx @@ -1,118 +1,102 @@ --- title: Change Data Capture -description: Implement CDC by streaming write-ahead log (WAL) SSTs with WalReader +description: Stream SlateDB WAL files with a live SlateDbWalReader iterator --- import { Code } from '@astrojs/starlight/components'; import cdcExample from '/../examples/src/change_data_capture.rs?raw'; -Change data capture (CDC) turns writes in SlateDB into a durable change stream that external systems can consume. SlateDB exposes CDC through the `WalReader`, which reads write-ahead log (WAL) SST files from object storage and returns the same row entries that were written by the database. +Change data capture (CDC) turns writes in SlateDB into a durable change stream that external systems can consume. SlateDB exposes native-WAL CDC through **slatedb::wal::SlateDbWalReader**. -This page explains how to build a CDC pipeline with `WalReader`, including ordering, durability, deletes, and resume semantics. +The reader opens a live iterator from the first unconsumed WAL file ID. The iterator owns polling and waits internally when it reaches the current WAL tail; the caller owns durable cursor storage. ## When to use WAL-based CDC Use WAL-based CDC when you need: - A low-latency stream of inserts, updates, and deletes. -- A simple, append-only feed that can be replayed. -- Integration with downstream systems like Kafka, Pulsar, or data lakes. +- A durable append-only feed that can be replayed. +- Integration with downstream systems such as Kafka, Pulsar, or data lakes. -WAL-based CDC is not a bootstrap or backfill API. If you need an initial full copy of data, combine a backfill with WAL tailing as described below. +WAL-based CDC is not a bootstrap or backfill API. Combine a database backfill with WAL streaming when an initial full copy is required. -## What WalReader reads +## Streaming model -SlateDB stores WAL SSTs under the `wal/` directory in the object store. Each WAL file has a monotonically increasing ID and is written in insertion order by sequence number. `WalReader` lists these files in ID order and exposes a `WalFile` for each file in object storage. +Keep a durable cursor containing the last fully consumed WAL file ID. Use zero when starting from the beginning unless you have a more appropriate manifest-derived cursor. -Each `WalFile` exposes an `iterator()` that yields `RowEntry` values in order. The iterator is optional because the underlying object might have been deleted between listing and reading (for example, due to GC). +To consume the stream: -Each row returned by the iterator is a `RowEntry` with: +1. Compute **cursor + 1**, checking for overflow, and create one iterator with an unbounded end range beginning at that WAL file ID. +2. Repeatedly call **next()**. When the iterator reaches the current tail, the call waits and polls internally for the next WAL file. +3. Emit every row in each **WalRows** batch. +4. After the whole batch succeeds, persist its **last_consumed_wal_file_id** as the new cursor. +5. After a restart, create a new unbounded iterator beginning at the persisted cursor plus one. -- `key`: the key bytes -- `value`: a `ValueDeletable` variant (`Value`, `Merge`, or `Tombstone`) -- `seq`: a globally increasing sequence number -- `create_ts` and `expire_ts`: optional timestamps +The live iterator does not report exhaustion at the current tail; **next()** remains pending until a new WAL file is visible or an error occurs. **last_wal_file_id** remains available when an application needs a point-in-time tail snapshot, but it is not part of the streaming loop. -Deletes appear as `Tombstone`s. If your database uses merge operators, you will see `Merge` values and must apply the merge logic yourself if your sink requires materialized values. +## Batches and cursor progress -:::note +One **WalRows** batch represents one fully consumed WAL file and contains: -The database must have a write-ahead log in order to use WAL-based CDC. The write-ahead log is controlled with `Settings::wal_enabled`. `wal_enabled` defaults to `true`, and is feature-gated behind a `wal_disable` feature in Rust. Databases have a WAL unless you have deliberaly disabled it. +- **rows**: the file's row entries, in sequence order +- **last_consumed_wal_file_id**: the durable cursor after that file -::: +Empty fence WAL files produce a batch with no rows. The cursor still advances, so consumers must process the batch rather than flattening batches into rows and losing progress information. -## Visibility and durability - -`WalReader` only sees WAL SSTs that have been flushed to object storage. WAL flushes occur when: - -- The WAL buffer reaches its size threshold ([`l0_sst_size_bytes`](https://docs.rs/slatedb/latest/slatedb/config/struct.Settings.html#structfield.l0_sst_size_bytes) is used to limit WAL SST file sizes as well). -- The configured [`flush_interval`](https://docs.rs/slatedb/latest/slatedb/config/struct.Settings.html#structfield.flush_interval) elapses. -- You explicitly call a WAL or MemTable flush with [`Db::flush_with_options`](https://docs.rs/slatedb/latest/slatedb/struct.Db.html#method.flush_with_options). - -Note that a memtable flush also forces any pending WAL data to be flushed first, which guarantees the WAL is durable before L0 data is written. - -## CDC architecture - -```mermaid -flowchart LR - A[Writer] --> B[WAL buffer] - B --> C[WAL SSTs in object store] - C --> D[WalReader] - D --> E[CDC sink] -``` +Each row is a **RowEntry** with: -`WalReader` does not interpret or compact data. It simply provides a durable, ordered feed of row-level changes. +- **key**: key bytes +- **value**: a **ValueDeletable** value, merge operand, or tombstone +- **seq**: the globally increasing sequence number +- **create_ts** and **expire_ts**: optional timestamps -## Basic tailer loop +Deletes appear as tombstones. Merge operands remain raw merge values; downstream consumers must apply the merge logic themselves when materialized values are required. -This example tails all WAL files, emits row entries, and records a cursor so the stream can resume after restarts. +## Visibility and durability - +SlateDbWalReader only sees WAL SSTs flushed to object storage. WAL flushes occur when: -This pattern is safe and idempotent as long as you persist the cursor after each emitted row and skip any rows with `seq <= last_seq` when resuming. +- The WAL buffer reaches its size threshold. +- The configured flush interval elapses. +- The caller explicitly requests a WAL or memtable flush. -## Backfills plus streaming +A memtable flush first forces pending WAL data to durable storage. -If you need a full backfill of data not in the WAL along with the WAL change stream: +The database must have its WAL enabled. WAL support is enabled by default. -1. Tail the WAL, buffering changes. -2. Checkpoint the database using `Admin::create_detached_checkpoint`. -3. Range scan (`Db::scan`) over the cloned database to read old data. -4. Apply the buffered WAL changes in sequence order, filtering rows whose sequence number was in the range scan in (3). -5. Delete the checkpoint using `Admin::delete_checkpoint`. -6. Continue tailing WALs from the stored cursor. +## Basic live stream -This avoids gaps between the backfill and the live stream. Your sink should be idempotent if the backfill and WAL stream can overlap. +This example creates one unbounded iterator, emits both existing and later writes through it, and advances a durable file cursor. -## Resumability and ordering + -`WalReader` lists files in wal ID order, and each WAL file stores entries in increasing sequence order. This gives you a total order that is stable across restarts. A minimal cursor includes: +Persisting the cursor only after a complete batch gives at-least-once delivery if a process fails while emitting that WAL file. Make the sink idempotent or transactionally couple sink writes with cursor persistence when duplicates are unacceptable. -- `wal_id`: the WAL file currently being processed -- `last_seq`: the highest sequence number processed in that file +## Ordering and resumability -If you need to fan out to multiple workers, partition by key and keep one cursor per partition. Ensure your sink can handle replays or duplicates. Since WAL files can be deleted between `list` and `iterator` calls, handle `None` by skipping that file or re-listing from a newer cursor. +WAL file IDs increase monotonically, and entries within each file are returned in sequence order. The minimal cursor is therefore the last fully consumed WAL file ID. -## Listing costs and polling strategy +Do not advance the cursor per row. Doing so can lose progress for empty fence WALs and can incorrectly mark a partially emitted WAL file as complete. -The `list()` API can become expensive when WAL retention is high or GC is not keeping up. If the GC is not running, listings can grow without bound. Even with GC, CDC often needs higher retention. Retaining WAL files for just 1 hour can yield tens of thousands of files, which is expensive to list in both cost (object-store listing calls) and time. +If a requested file has already been garbage collected, iteration reports a truncation/data error. The consumer must not silently skip the gap. -If you plan to poll frequently: +## Backfills plus streaming -1. Use `list()` once to get an initial view (or to recover after a long outage). -2. Track the highest WAL ID you have successfully processed. -3. From then on, poll using `WalReader::get(latest_id + 1)`. +For a full backfill: -## Deletes, merges, and TTL +1. Begin live WAL streaming and buffer changes. +2. Create a detached database checkpoint. +3. Scan the checkpoint or a clone for existing data. +4. Apply buffered WAL changes in sequence order. +5. Delete the checkpoint. +6. Continue streaming from the persisted WAL cursor. -- Deletes appear as `ValueDeletable::Tombstone`. Emit a delete event in your CDC sink. -- Merge operands appear as `ValueDeletable::Merge`. If you use merge operators, downstream consumers must either apply the merge or store the operand as-is. -- TTL is represented by `expire_ts`. If your downstream system enforces TTL, honor this field. +The sink should tolerate overlap between the backfill and WAL stream. ## WAL retention and GC -WAL SSTs are garbage collected based on the GC configuration. If your CDC consumer falls behind, the WAL files it needs may be deleted. Set the WAL GC `min_age` option to retain files long enough for your slowest consumer and implement monitoring. Slow consumers should copy the WAL files elsewhere for processing. See [Garbage Collection](/docs/design/gc) for details. +WAL SSTs are garbage collected according to the configured retention policy. Set WAL GC minimum age long enough for the slowest consumer and monitor cursor lag. A consumer that requests a deleted WAL receives a truncation error instead of silently continuing. -## Using a separate WAL object store +## Dedicated WAL object stores -If your database uses a dedicated WAL object store, pass that store to `WalReader::new` rather than the main store. +When a database uses a dedicated WAL object store, construct the reader with both stores: the main object store is used for manifest and GC-cutoff state, and the dedicated WAL object store is used for WAL SSTs. In Rust, use **SlateDbWalReader::new_for_db_with_wal_object_store**. From e0161973d8d7ffdede7c44725729838811674e99 Mon Sep 17 00:00:00 2001 From: Rohan Date: Sat, 22 Aug 2026 13:16:10 -0400 Subject: [PATCH 38/65] configure sst iterator read ahead using bytes (#2037) --- slatedb/src/compactor_executor.rs | 4 +- slatedb/src/config.rs | 14 +- slatedb/src/format/sst.rs | 8 +- slatedb/src/reader.rs | 6 +- slatedb/src/sst_iter.rs | 61 ++++---- slatedb/src/tablestore.rs | 194 ++++++++++++++++++++++--- slatedb/src/wal/slatedb/reader.rs | 9 +- slatedb/src/wal/slatedb/writer_init.rs | 2 +- slatedb/src/wal_reader.rs | 4 +- 9 files changed, 223 insertions(+), 79 deletions(-) diff --git a/slatedb/src/compactor_executor.rs b/slatedb/src/compactor_executor.rs index 9fff4281e7..926f609e48 100644 --- a/slatedb/src/compactor_executor.rs +++ b/slatedb/src/compactor_executor.rs @@ -338,9 +338,7 @@ impl TokioCompactionExecutorInner { }; let sst_iter_options = SstIteratorOptions { max_fetch_tasks: self.options.max_fetch_tasks, - blocks_to_fetch: self - .table_store - .bytes_to_blocks(self.options.bytes_to_fetch), + target_bytes_to_fetch: self.options.bytes_to_fetch, cache_blocks: false, // don't clobber the cache cache_metadata: false, eager_spawn: true, diff --git a/slatedb/src/config.rs b/slatedb/src/config.rs index 27339db7ab..c595134096 100644 --- a/slatedb/src/config.rs +++ b/slatedb/src/config.rs @@ -350,9 +350,10 @@ pub struct ScanOptions { /// Whether to include dirty data in the scan. "dirty" means that the data is not considered /// as "committed" yet, whose seq number is greater than the last committed seq number. pub dirty: bool, - /// The number of bytes to read ahead. The value is rounded up to the nearest - /// block size when fetching from object storage. The default is 1, which - /// rounds up to one block. + /// The target number of bytes to fetch in a single request while iterating over SSTs. + /// Each fetch will read the minimum number of blocks such that the resulting read is at least + /// this size or reaches the end of the file. The default is 1, which results in each fetch + /// reading one block. pub read_ahead_bytes: usize, /// Whether or not fetched data blocks should be cached. SST indexes, /// filters, and stats are cached independently of this setting. @@ -1321,9 +1322,10 @@ pub struct CompactionWorkerOptions { /// compaction. Higher values can improve throughput but use more resources. pub max_fetch_tasks: usize, - /// Number of bytes to fetch in a single read-ahead request while iterating - /// over input SSTs during compaction. The value is rounded up to the nearest - /// block size when fetching from object storage. The default is 2MiB. + /// The target number of bytes to fetch in a single request while iterating over + /// input SSTs during compaction. Each fetch will read the minimum number of blocks + /// such that the resulting read is at least this size or reaches the end of the file. + /// The default is 2MiB. /// /// This pairs with [`CompactionWorkerOptions::max_fetch_tasks`]: /// `bytes_to_fetch` is the size of each read-ahead request while diff --git a/slatedb/src/format/sst.rs b/slatedb/src/format/sst.rs index 7413c98615..c0171838b0 100644 --- a/slatedb/src/format/sst.rs +++ b/slatedb/src/format/sst.rs @@ -916,13 +916,17 @@ impl SsTableFormat { } } - fn block_range( + pub(crate) fn block_range( &self, blocks: Range, info: &SsTableInfo, index: &SsTableIndex, ) -> Range { - let mut end_offset = info.filter_offset; + let mut end_offset = if info.filter_len > 0 { + info.filter_offset + } else { + info.index_offset + }; if blocks.end < index.block_meta().len() { let next_block_meta = index.block_meta().get(blocks.end); end_offset = next_block_meta.offset(); diff --git a/slatedb/src/reader.rs b/slatedb/src/reader.rs index 1187c0c7c5..9049e7e1aa 100644 --- a/slatedb/src/reader.rs +++ b/slatedb/src/reader.rs @@ -280,11 +280,10 @@ impl Reader { ) -> Result { self.db_stats.scan_requests.increment(1); let max_seq = self.prepare_max_seq(ctx.max_seq, options.durability_filter, options.dirty); - let read_ahead_blocks = self.table_store.bytes_to_blocks(options.read_ahead_bytes); let sst_iter_options = SstIteratorOptions { max_fetch_tasks: options.max_fetch_tasks, - blocks_to_fetch: read_ahead_blocks, + target_bytes_to_fetch: options.read_ahead_bytes, cache_blocks: options.cache_blocks, cache_metadata: true, eager_spawn: true, @@ -342,12 +341,11 @@ impl Reader { ) -> Result { self.db_stats.scan_requests.increment(1); let max_seq = self.prepare_max_seq(None, options.durability_filter, options.dirty); - let read_ahead_blocks = self.table_store.bytes_to_blocks(options.read_ahead_bytes); let range = BytesRange::from_prefix(prefix.as_ref()); let sst_iter_options = SstIteratorOptions { max_fetch_tasks: options.max_fetch_tasks, - blocks_to_fetch: read_ahead_blocks, + target_bytes_to_fetch: options.read_ahead_bytes, cache_blocks: options.cache_blocks, cache_metadata: true, // Recency scans are designed for early-stop. Eager spawning would diff --git a/slatedb/src/sst_iter.rs b/slatedb/src/sst_iter.rs index edee72f77e..8bd7f547f7 100644 --- a/slatedb/src/sst_iter.rs +++ b/slatedb/src/sst_iter.rs @@ -2,7 +2,6 @@ use async_trait::async_trait; use bytes::Bytes; use log::error; use slatedb_common::metrics::CounterFn; -use std::cmp::min; use std::collections::VecDeque; use std::ops::Bound::{Excluded, Included, Unbounded}; use std::ops::{Bound, Range, RangeBounds}; @@ -34,7 +33,7 @@ enum FetchTask { #[derive(Clone, Debug)] pub(crate) struct SstIteratorOptions { pub(crate) max_fetch_tasks: usize, - pub(crate) blocks_to_fetch: usize, + pub(crate) target_bytes_to_fetch: usize, pub(crate) cache_blocks: bool, pub(crate) cache_metadata: bool, pub(crate) eager_spawn: bool, @@ -47,7 +46,7 @@ impl Default for SstIteratorOptions { fn default() -> Self { SstIteratorOptions { max_fetch_tasks: 1, - blocks_to_fetch: 1, + target_bytes_to_fetch: 1, cache_blocks: true, cache_metadata: true, eager_spawn: false, @@ -293,7 +292,7 @@ impl<'a> InternalSstIterator<'a> { options: SstIteratorOptions, ) -> Result { assert!(options.max_fetch_tasks > 0); - assert!(options.blocks_to_fetch > 0); + assert!(options.target_bytes_to_fetch > 0); let descending_buffer = match options.order { IterationOrder::Descending => Some(VecDeque::new()), @@ -381,25 +380,23 @@ impl<'a> InternalSstIterator<'a> { while self.fetch_tasks.len() < self.options.max_fetch_tasks && self.block_idx_range.contains(&self.next_block_idx_to_fetch) { - let blocks_to_fetch = min( - self.options.blocks_to_fetch, - self.block_idx_range.end - self.next_block_idx_to_fetch, - ); let table = self.view.table_as_ref().sst.clone(); + let mut blocks = self.table_store.block_range_for_target_bytes( + &table, + index, + self.next_block_idx_to_fetch, + self.options.target_bytes_to_fetch, + IterationOrder::Ascending, + ); + blocks.end = blocks.end.min(self.block_idx_range.end); let table_store = self.table_store.clone(); - let blocks_start = self.next_block_idx_to_fetch; - let blocks_end = self.next_block_idx_to_fetch + blocks_to_fetch; let index = index.clone(); let cache_blocks = self.options.cache_blocks; + let blocks_end = blocks.end; self.fetch_tasks .push_back(FetchTask::InFlight(tokio::spawn(async move { table_store - .read_blocks_using_index( - &table, - index, - blocks_start..blocks_end, - cache_blocks, - ) + .read_blocks_using_index(&table, index, blocks, cache_blocks) .await }))); self.next_block_idx_to_fetch = blocks_end; @@ -410,25 +407,23 @@ impl<'a> InternalSstIterator<'a> { while self.fetch_tasks.len() < self.options.max_fetch_tasks && self.next_block_idx_to_fetch > self.block_idx_range.start { - let blocks_to_fetch = min( - self.options.blocks_to_fetch, - self.next_block_idx_to_fetch - self.block_idx_range.start, - ); let table = self.view.table_as_ref().sst.clone(); + let mut blocks = self.table_store.block_range_for_target_bytes( + &table, + index, + self.next_block_idx_to_fetch - 1, + self.options.target_bytes_to_fetch, + IterationOrder::Descending, + ); + blocks.start = blocks.start.max(self.block_idx_range.start); let table_store = self.table_store.clone(); - let blocks_end = self.next_block_idx_to_fetch; - let blocks_start = blocks_end - blocks_to_fetch; let index = index.clone(); let cache_blocks = self.options.cache_blocks; + let blocks_start = blocks.start; self.fetch_tasks .push_back(FetchTask::InFlight(tokio::spawn(async move { table_store - .read_blocks_using_index( - &table, - index, - blocks_start..blocks_end, - cache_blocks, - ) + .read_blocks_using_index(&table, index, blocks, cache_blocks) .await }))); self.next_block_idx_to_fetch = blocks_start; @@ -1616,7 +1611,7 @@ mod tests { let sst_iter_options = SstIteratorOptions { max_fetch_tasks: 3, - blocks_to_fetch: 3, + target_bytes_to_fetch: 3 * 4096, cache_blocks: true, order, ..SstIteratorOptions::default() @@ -1900,7 +1895,7 @@ mod tests { table_store.clone(), SstIteratorOptions { max_fetch_tasks: 32, - blocks_to_fetch: 256, + target_bytes_to_fetch: 256 * 128, cache_blocks: true, cache_metadata: true, eager_spawn: false, @@ -1919,7 +1914,7 @@ mod tests { table_store.clone(), SstIteratorOptions { max_fetch_tasks: 1, - blocks_to_fetch: 1, + target_bytes_to_fetch: 1, cache_blocks: true, cache_metadata: true, eager_spawn: false, @@ -2506,7 +2501,7 @@ mod tests { let sst_iter_options = SstIteratorOptions { max_fetch_tasks: 3, - blocks_to_fetch: 3, + target_bytes_to_fetch: 3 * 128, cache_blocks: true, cache_metadata: true, eager_spawn: false, @@ -2799,7 +2794,7 @@ mod tests { table_store.clone(), SstIteratorOptions { max_fetch_tasks: 1, - blocks_to_fetch: 1, + target_bytes_to_fetch: 1, cache_blocks: true, cache_metadata: true, eager_spawn: false, diff --git a/slatedb/src/tablestore.rs b/slatedb/src/tablestore.rs index e496824743..4d1b6bbf4d 100644 --- a/slatedb/src/tablestore.rs +++ b/slatedb/src/tablestore.rs @@ -26,6 +26,7 @@ use crate::filter_policy::NamedFilter; use crate::flatbuffer_types::SsTableIndexOwned; use crate::format::block::Block; use crate::format::sst::{EncodedSsTable, EncodedSsTableBlock, SsTableFormat}; +use crate::iter::IterationOrder; use crate::object_store_tag::ObjectStoreCallTag; pub(crate) use crate::object_store_tag::TableStoreKind; use crate::object_stores::{ObjectStoreType, ObjectStores}; @@ -161,12 +162,6 @@ impl TableStore { } } - /// Get the number of blocks for a size specified in bytes. - /// The returned value will be rounded down to the nearest block. - pub(crate) fn bytes_to_blocks(&self, bytes: usize) -> usize { - bytes.div_ceil(self.sst_format.block_size) - } - /// Find the highest WAL SST id present in the object store at or above /// `start_after + 1`, returning `start_after` if none exist. /// @@ -870,6 +865,57 @@ impl TableStore { .await } + /// Returns the smallest contiguous block range, starting at `first_block` in + /// `order`, whose encoded size is at least `target_bytes`. If the SST boundary + /// is reached first, all remaining blocks in that direction are returned. + pub(crate) fn block_range_for_target_bytes( + &self, + handle: &SsTableHandle, + index: &SsTableIndexOwned, + first_block: usize, + target_bytes: usize, + order: IterationOrder, + ) -> Range { + assert!(target_bytes > 0); + + let index = index.borrow(); + let block_meta = index.block_meta(); + let num_blocks = block_meta.len(); + assert!(first_block < num_blocks); + + let target_bytes = u64::try_from(target_bytes).unwrap_or(u64::MAX); + match order { + IterationOrder::Ascending => { + let mut blocks = first_block..first_block + 1; + loop { + let byte_range = + self.sst_format + .block_range(blocks.clone(), &handle.info, &index); + if byte_range.end.saturating_sub(byte_range.start) >= target_bytes + || blocks.end == num_blocks + { + return blocks; + } + blocks.end += 1; + } + } + IterationOrder::Descending => { + let mut blocks = first_block..first_block + 1; + loop { + let byte_range = + self.sst_format + .block_range(blocks.clone(), &handle.info, &index); + if byte_range.end.saturating_sub(byte_range.start) >= target_bytes + || blocks.start == 0 + { + return blocks; + } + blocks.start -= 1; + } + } + } + } + /// Reads specified blocks from an SSTable using the provided index. /// /// This function attempts to read blocks from the cache if available @@ -1311,10 +1357,9 @@ mod tests { use futures::future; use futures::StreamExt; use object_store::{memory::InMemory, path::Path, ObjectStore, ObjectStoreExt}; - use proptest::prelude::any; - use proptest::proptest; use rstest::rstest; use std::collections::VecDeque; + use std::ops::Range; use std::sync::Arc; use crate::block_cache_policy::BlockCachePolicy; @@ -1322,9 +1367,14 @@ mod tests { use crate::db_cache::CacheTarget; use crate::db_cache::SplitCache; use crate::db_cache::{CachedKey, DbCache, DbCacheWrapper}; + use crate::db_state::{SsTableHandle, SsTableInfo}; use crate::error; + use crate::flatbuffer_types::{ + BlockMeta, BlockMetaArgs, SsTableIndex, SsTableIndexArgs, SsTableIndexOwned, + }; use crate::format::block::Block; - use crate::format::sst::SsTableFormat; + use crate::format::sst::{SsTableFormat, SST_FORMAT_VERSION_LATEST}; + use crate::iter::IterationOrder; use crate::manifest::SsTableView; use crate::object_stores::ObjectStores; use crate::retrying_object_store::RetryingObjectStore; @@ -1431,6 +1481,33 @@ mod tests { Arc::new(InMemory::new()) } + fn build_index(offsets: &[u64]) -> SsTableIndexOwned { + let mut builder = flatbuffers::FlatBufferBuilder::new(); + let block_meta = offsets + .iter() + .enumerate() + .map(|(block, offset)| { + let first_key = builder.create_vector(block.to_string().as_bytes()); + BlockMeta::create( + &mut builder, + &BlockMetaArgs { + offset: *offset, + first_key: Some(first_key), + }, + ) + }) + .collect::>(); + let block_meta = builder.create_vector(&block_meta); + let index = SsTableIndex::create( + &mut builder, + &SsTableIndexArgs { + block_meta: Some(block_meta), + }, + ); + builder.finish(index, None); + SsTableIndexOwned::new(Bytes::copy_from_slice(builder.finished_data())).unwrap() + } + async fn count_ssts_in(store: &Arc) -> usize { store .list(None) @@ -2926,20 +3003,91 @@ mod tests { assert_eq!(metadata.location, path); } - proptest! { - #[test] - fn convert_bytes_to_blocks_precise_when_aligned_with_block_size( - block_size in any::(), - num_blocks in any::(), - ) { - let os = Arc::new(InMemory::new()); - let format = SsTableFormat { block_size, ..SsTableFormat::default() }; - let ts = Arc::new(TableStore::new(ObjectStores::new(os, None), - format, Path::from(ROOT), None, TableStoreKind::Main, BlockCachePolicy::default())); - if let Some(bytes) = block_size.checked_mul(num_blocks) { - assert_eq!(num_blocks, ts.bytes_to_blocks(bytes)); - } - } + #[rstest] + #[case::ascending_one_block(IterationOrder::Ascending, 0, 100, 0..1)] + #[case::ascending_crosses_boundary(IterationOrder::Ascending, 0, 101, 0..2)] + #[case::ascending_exact_boundary(IterationOrder::Ascending, 0, 250, 0..2)] + #[case::ascending_exhausts_sst(IterationOrder::Ascending, 1, 1_000, 1..4)] + #[case::descending_one_block(IterationOrder::Descending, 3, 200, 3..4)] + #[case::descending_crosses_boundary(IterationOrder::Descending, 3, 201, 2..4)] + #[case::descending_exact_boundary(IterationOrder::Descending, 3, 250, 2..4)] + #[case::descending_exhausts_sst(IterationOrder::Descending, 2, 1_000, 0..3)] + fn block_range_for_target_bytes_is_minimal( + #[case] order: IterationOrder, + #[case] first_block: usize, + #[case] target_bytes: usize, + #[case] expected: Range, + ) { + let table_store = TableStore::new( + ObjectStores::new(make_store(), None), + SsTableFormat::default(), + Path::from(ROOT), + None, + TableStoreKind::Main, + BlockCachePolicy::default(), + ); + let handle = SsTableHandle::new( + SsTableId::Compacted(ulid::Ulid::new()), + SST_FORMAT_VERSION_LATEST, + SsTableInfo { + index_offset: 500, + filter_offset: 500, + ..SsTableInfo::default() + }, + ); + // Encoded block sizes are 100, 150, 50, and 200 bytes. + let index = build_index(&[0, 100, 250, 300]); + + let actual = table_store.block_range_for_target_bytes( + &handle, + &index, + first_block, + target_bytes, + order, + ); + + assert_eq!(actual, expected); + } + + #[rstest] + #[case::filter_precedes_index(1, 500, 700)] + #[case::index_follows_data_without_filter(0, 700, 500)] + fn block_range_for_target_bytes_uses_format_for_last_block_end( + #[case] filter_len: u64, + #[case] filter_offset: u64, + #[case] index_offset: u64, + ) { + let table_store = TableStore::new( + ObjectStores::new(make_store(), None), + SsTableFormat::default(), + Path::from(ROOT), + None, + TableStoreKind::Main, + BlockCachePolicy::default(), + ); + let handle = SsTableHandle::new( + SsTableId::Compacted(ulid::Ulid::new()), + SST_FORMAT_VERSION_LATEST, + SsTableInfo { + index_offset, + filter_offset, + filter_len, + ..SsTableInfo::default() + }, + ); + let index = build_index(&[0, 100, 250, 300]); + + let actual = table_store.block_range_for_target_bytes( + &handle, + &index, + 3, + 201, + IterationOrder::Descending, + ); + + // The last block ends at byte 500 and is 200 bytes, so one preceding + // block is required to meet a 201-byte target. + assert_eq!(actual, 2..4); } /// End-to-end test: concurrent index reads through `TableStore` issue a single diff --git a/slatedb/src/wal/slatedb/reader.rs b/slatedb/src/wal/slatedb/reader.rs index 9e53cf00fe..f8c7b246ed 100644 --- a/slatedb/src/wal/slatedb/reader.rs +++ b/slatedb/src/wal/slatedb/reader.rs @@ -30,8 +30,9 @@ pub struct SlateDbWalReaderOptions { /// pending while the current data is being consumed. pub max_fetch_tasks: usize, - /// The number of bytes to read ahead in each sst. The value is rounded up to the nearest - /// block size when fetching from object storage. The default is 1MB + /// The target number of bytes to fetch in a single request while iterating over WAL SSTs. + /// Each fetch will read the minimum number of blocks such that the resulting read is at least + /// this size or reaches the end of the file. The default is 1MiB. pub read_ahead_bytes: usize, } @@ -47,13 +48,11 @@ impl Default for SlateDbWalReaderOptions { impl From for SlateDbWalIteratorOptions { fn from(options: SlateDbWalReaderOptions) -> Self { - let format = SsTableFormat::default(); - let blocks_to_fetch = options.read_ahead_bytes.div_ceil(format.block_size); Self { sst_batch_size: options.sst_batch_size, sst_iter_options: SstIteratorOptions { max_fetch_tasks: options.max_fetch_tasks, - blocks_to_fetch, + target_bytes_to_fetch: options.read_ahead_bytes, cache_blocks: false, cache_metadata: false, eager_spawn: true, diff --git a/slatedb/src/wal/slatedb/writer_init.rs b/slatedb/src/wal/slatedb/writer_init.rs index 59404802ba..35b93051f8 100644 --- a/slatedb/src/wal/slatedb/writer_init.rs +++ b/slatedb/src/wal/slatedb/writer_init.rs @@ -125,7 +125,7 @@ impl wal::WriterInit for SlateDbWalWriterInit { sst_batch_size: 4, sst_iter_options: SstIteratorOptions { max_fetch_tasks: 2, - blocks_to_fetch: 256, + target_bytes_to_fetch: 1024 * 1024, cache_blocks: false, cache_metadata: false, eager_spawn: true, diff --git a/slatedb/src/wal_reader.rs b/slatedb/src/wal_reader.rs index cdd3eb2181..a70e41b873 100644 --- a/slatedb/src/wal_reader.rs +++ b/slatedb/src/wal_reader.rs @@ -153,8 +153,8 @@ impl WalFile { SsTableView::identity(sst), Arc::clone(&self.table_store), SstIteratorOptions { - // Optimize for throughput. Go for 256MiB per-fetch (4096 bytes/block default). - blocks_to_fetch: 65_536, + // Optimize for throughput. Go for 256MiB per fetch. + target_bytes_to_fetch: 256 * 1024 * 1024, ..Default::default() }, ) From 163359d4d93e1621071b9ae2d72fe5752d920cd2 Mon Sep 17 00:00:00 2001 From: Aditya Mishra Date: Tue, 25 Aug 2026 00:59:01 +0530 Subject: [PATCH 39/65] perf(wal): read a whole WAL SST per request during replay (#2003) (#2042) --- slatedb/src/db_reader.rs | 54 +++++++++++++++++++++++++- slatedb/src/wal/slatedb/reader.rs | 7 ++-- slatedb/src/wal/slatedb/writer_init.rs | 25 ++++-------- 3 files changed, 64 insertions(+), 22 deletions(-) diff --git a/slatedb/src/db_reader.rs b/slatedb/src/db_reader.rs index 640fa2a603..2716e44ffd 100644 --- a/slatedb/src/db_reader.rs +++ b/slatedb/src/db_reader.rs @@ -204,7 +204,10 @@ impl DbReaderInner { Arc::clone(&table_store), &status_manager, Arc::clone(&system_clock), - SlateDbWalReaderOptions::default(), + SlateDbWalReaderOptions { + read_ahead_bytes: options.max_memtable_bytes as usize, + ..SlateDbWalReaderOptions::default() + }, ), ) }); @@ -2976,6 +2979,55 @@ mod tests { ); } + /// Regression test for #2003: read-ahead was a fixed 1MiB, so a large WAL took one + /// GET per MiB. It now covers a whole WAL SST, so reads for one file stay a small + /// constant instead of growing with size. + #[tokio::test] + async fn replay_reads_a_large_wal_sst_in_a_bounded_number_of_requests() { + let recording_store = Arc::new(test_utils::RecordingObjectStore::new(Arc::new( + InMemory::new(), + ))); + let object_store: Arc = recording_store.clone(); + let path = Path::from("/tmp/test_kv_store"); + let test_provider = TestProvider::new(path.clone(), Arc::clone(&object_store)); + let table_store = test_provider.table_store(); + + // One 16MiB WAL SST, far over the old 1MiB window. The old window read it in + // ~16 data GETs; the fix reads it in one. + let value = vec![b'x'; 4096]; + let entries: Vec = (0..4096u32) + .map(|i| RowEntry::new_value(format!("key-{i:08}").as_bytes(), &value, i as u64 + 1)) + .collect(); + write_wal_sst(Arc::clone(&table_store), 1, entries) + .await + .unwrap(); + + let mut core = ManifestCore::new(); + core.next_wal_sst_id = 2; + let status_manager = status_manager_for_core(&core); + let wal_reader = native_wal_reader(&table_store, &status_manager); + + recording_store.clear(); + let mut iterator = wal_reader.iterator((1..2).into()).await.unwrap(); + let mut rows = 0; + while let Some(batch) = iterator.next().await.unwrap() { + rows += batch.rows.len(); + } + assert_eq!(rows, 4096, "replay should return every WAL row"); + + let wal_reads = recording_store + .get_sst_types(false) + .into_iter() + .filter(|sst_type| *sst_type == Some(SstType::Wal)) + .count(); + // Footer, index, and one data read. The old 1MiB window needed ~16 data reads + // for this file, so the bound is the regression. + assert!( + wal_reads <= 4, + "expected a bounded number of WAL reads for one file, got {wal_reads}" + ); + } + /// A checkpoint captures the WAL files that were durable when it was taken, /// so a pinned reader replays them by default. `skip_wal_replay` opts out of /// that read, at the cost of not seeing the checkpointed WAL writes. diff --git a/slatedb/src/wal/slatedb/reader.rs b/slatedb/src/wal/slatedb/reader.rs index f8c7b246ed..e0503598a8 100644 --- a/slatedb/src/wal/slatedb/reader.rs +++ b/slatedb/src/wal/slatedb/reader.rs @@ -31,8 +31,9 @@ pub struct SlateDbWalReaderOptions { pub max_fetch_tasks: usize, /// The target number of bytes to fetch in a single request while iterating over WAL SSTs. - /// Each fetch will read the minimum number of blocks such that the resulting read is at least - /// this size or reaches the end of the file. The default is 1MiB. + /// Each fetch reads the minimum number of blocks such that the resulting read is at least + /// this size or reaches the end of the file. Callers size this to a whole WAL SST + /// (`l0_sst_size_bytes`) so replay reads each file in one request. pub read_ahead_bytes: usize, } @@ -41,7 +42,7 @@ impl Default for SlateDbWalReaderOptions { Self { sst_batch_size: 4, max_fetch_tasks: 2, - read_ahead_bytes: 1024 * 1024, + read_ahead_bytes: 64 * 1024 * 1024, } } } diff --git a/slatedb/src/wal/slatedb/writer_init.rs b/slatedb/src/wal/slatedb/writer_init.rs index 35b93051f8..84a3dc7042 100644 --- a/slatedb/src/wal/slatedb/writer_init.rs +++ b/slatedb/src/wal/slatedb/writer_init.rs @@ -1,13 +1,10 @@ use crate::dispatcher::MessageHandlerExecutor; use crate::error::SlateDBError; -use crate::iter::IterationOrder; use crate::manifest::Manifest; -use crate::sst_iter::SstIteratorOptions; use crate::tablestore::TableStore; use crate::utils::WatchableOnceCellReader; -use crate::wal::slatedb::iterator::{ - SlateDbWalIterator, SlateDbWalIteratorOptions, WalIteratorEndBound, -}; +use crate::wal::slatedb::iterator::{SlateDbWalIterator, WalIteratorEndBound}; +use crate::wal::slatedb::reader::SlateDbWalReaderOptions; use crate::wal::slatedb::writer::SlateDbWalWriter; use crate::wal::{WalError, WriterInitResult, WriterManifest}; use crate::{wal, Settings}; @@ -121,19 +118,11 @@ impl wal::WriterInit for SlateDbWalWriterInit { let replay_iterator = SlateDbWalIterator::range( replay_after_wal_id + 1, WalIteratorEndBound::Exclusive(empty_wal_id + 1), - SlateDbWalIteratorOptions { - sst_batch_size: 4, - sst_iter_options: SstIteratorOptions { - max_fetch_tasks: 2, - target_bytes_to_fetch: 1024 * 1024, - cache_blocks: false, - cache_metadata: false, - eager_spawn: true, - order: IterationOrder::Ascending, - prefix: None, - filter_context: None, - }, - }, + SlateDbWalReaderOptions { + read_ahead_bytes: self.max_wal_bytes_size, + ..SlateDbWalReaderOptions::default() + } + .into(), self.table_store.clone(), )?; let wal_writer = SlateDbWalWriter::start_new( From 84775a817c7e47384c8759fa8d4733bc4c13b4b7 Mon Sep 17 00:00:00 2001 From: mdwaud Date: Mon, 24 Aug 2026 15:51:39 -0400 Subject: [PATCH 40/65] Remove SQLync from Adopters (#2045) --- README.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/README.md b/README.md index 63870960d0..3cef5e7549 100644 --- a/README.md +++ b/README.md @@ -152,6 +152,7 @@ SlateDB follows Semantic Versioning. We release new versions approximately every See who's using SlateDB. +- [4og.io](https://4og.io) - [Dropbox](https://www.dropbox.com) - [Embucket](https://www.embucket.com) - [Gadget](https://gadget.dev) @@ -164,7 +165,6 @@ See who's using SlateDB. - [Prisma](https://www.prisma.io) - [Responsive](https://responsive.dev) - [s2-lite](https://github.com/s2-streamstore/s2) -- [SQLync](https://sqlync.com) - [Storrito](https://storrito.com) - [Taquba](https://github.com/micllam/taquba) - [Tensorlake](https://www.tensorlake.ai) From dd5a227faf1f142a0f0457a6b9eb99ba4dd413a8 Mon Sep 17 00:00:00 2001 From: Efrat Levitan <41479945+Efrat19@users.noreply.github.com> Date: Tue, 25 Aug 2026 18:50:18 +0300 Subject: [PATCH 41/65] Tolerate manifests GCed during admin.list_manifests (#2047) --- slatedb/src/admin.rs | 76 ++++++++++++++++++++++++++++++++++++--- slatedb/src/test_utils.rs | 26 ++++++++++++++ 2 files changed, 98 insertions(+), 4 deletions(-) diff --git a/slatedb/src/admin.rs b/slatedb/src/admin.rs index 81ba412d71..040705c9e5 100644 --- a/slatedb/src/admin.rs +++ b/slatedb/src/admin.rs @@ -107,11 +107,21 @@ impl Admin { .map_err(crate::Error::from)?; let mut manifests = Vec::with_capacity(manifest_metadata.len()); for metadata in manifest_metadata { - let manifest = manifest_store - .read_manifest(metadata.id) + match manifest_store + .try_read_manifest(metadata.id) .await - .map_err(crate::Error::from)?; - manifests.push(VersionedManifest::from_manifest(metadata.id, manifest)); + .map_err(crate::Error::from)? + { + Some(manifest) => { + manifests.push(VersionedManifest::from_manifest(metadata.id, manifest)) + } + // Deleted after LIST by a concurrent GC + // See https://github.com/slatedb/slatedb/issues/1215 for more details. + None => log::warn!( + "listed manifest missing on read, skipping [id={}]", + metadata.id + ), + } } Ok(manifests) } @@ -1978,3 +1988,61 @@ mod tests { } } } + +#[cfg(test)] +mod gc_tolerant_list_manifests_tests { + use crate::admin::AdminBuilder; + use crate::manifest::store::{ManifestStore, StoredManifest}; + use crate::manifest::ManifestCore; + use crate::test_utils::FlakyObjectStore; + use object_store::memory::InMemory; + use object_store::path::Path; + use object_store::ObjectStore; + use slatedb_common::clock::DefaultSystemClock; + use std::sync::Arc; + + #[tokio::test] + async fn test_list_manifests_skips_manifest_gced_between_list_and_read() { + let inner: Arc = Arc::new(InMemory::new()); + let flaky = Arc::new(FlakyObjectStore::new(inner, 0)); + let store: Arc = flaky.clone(); + let path = Path::from("/tmp/test_gc_tolerant_list_manifests"); + + let manifest_store = Arc::new(ManifestStore::new(&path, store.clone())); + let mut sm = StoredManifest::create_new_db( + manifest_store, + ManifestCore::new(), + Arc::new(DefaultSystemClock::new()), + ) + .await + .unwrap(); + sm.update(sm.prepare_dirty().unwrap()).await.unwrap(); + sm.update(sm.prepare_dirty().unwrap()).await.unwrap(); + + let admin = AdminBuilder::new(path.clone(), store.clone()).build(); + let ids: Vec = admin + .list_manifests(..) + .await + .unwrap() + .iter() + .map(|vm| vm.id()) + .collect(); + assert_eq!(ids, vec![1, 2, 3]); + + // manifest 1 is listed, but missing on read + flaky.with_get_not_found_failures(1); + + let ids: Vec = admin + .list_manifests(..) + .await + .unwrap() + .iter() + .map(|vm| vm.id()) + .collect(); + assert_eq!( + ids, + vec![2, 3], + "a manifest GC'd mid-listing should be skipped, not fail the operation" + ); + } +} diff --git a/slatedb/src/test_utils.rs b/slatedb/src/test_utils.rs index 6be520b303..d730abfd71 100644 --- a/slatedb/src/test_utils.rs +++ b/slatedb/src/test_utils.rs @@ -573,6 +573,8 @@ pub(crate) struct FlakyObjectStore { // get_range: truncate response body to this many bytes on first N attempts (0 = no truncation) truncate_get_range_bytes: AtomicUsize, truncate_get_range_count: AtomicUsize, + // Get: return NotFound on the next N GETs + fail_first_get_not_found: AtomicUsize, } impl FlakyObjectStore { @@ -598,9 +600,14 @@ impl FlakyObjectStore { get_range_attempts: AtomicUsize::new(0), truncate_get_range_bytes: AtomicUsize::new(0), truncate_get_range_count: AtomicUsize::new(0), + fail_first_get_not_found: AtomicUsize::new(0), } } + pub(crate) fn with_get_not_found_failures(&self, n: usize) { + self.fail_first_get_not_found.store(n, Ordering::SeqCst); + } + pub(crate) fn with_put_precondition_always(self) -> Self { self.put_precondition_always.store(true, Ordering::SeqCst); self @@ -726,6 +733,25 @@ impl ObjectStore for FlakyObjectStore { location: &Path, options: GetOptions, ) -> object_store::Result { + if self + .fail_first_get_not_found + .fetch_update(Ordering::SeqCst, Ordering::SeqCst, |v| { + if v > 0 { + Some(v - 1) + } else { + None + } + }) + .is_ok() + { + return Err(object_store::Error::NotFound { + path: location.to_string(), + source: Box::new(std::io::Error::new( + std::io::ErrorKind::NotFound, + "injected not-found (deleted between LIST and GET)", + )), + }); + } if options.head { self.head_attempts.fetch_add(1, Ordering::SeqCst); if self From a6f0522a7a3b1120061f325bcb475eab9b45f2b3 Mon Sep 17 00:00:00 2001 From: Samuel Stroschein <35429197+samuelstroschein@users.noreply.github.com> Date: Tue, 25 Aug 2026 10:39:33 -0700 Subject: [PATCH 42/65] add lixray as adopter (#2050) --- README.md | 1 + 1 file changed, 1 insertion(+) diff --git a/README.md b/README.md index 3cef5e7549..54ce988b10 100644 --- a/README.md +++ b/README.md @@ -171,6 +171,7 @@ See who's using SlateDB. - [Volga](https://github.com/volga-project/volga) - [WombatKV](https://github.com/Venkat2811/wombatkv) - [ZeroFS](https://zerofs.net) +- [LixRay](https://lixray.com) ## Talks From 5d0da8da369fab8f808efd7288a6015dc77fb166 Mon Sep 17 00:00:00 2001 From: Bruno Cadonna Date: Tue, 25 Aug 2026 19:42:21 +0200 Subject: [PATCH 43/65] Propose tracing instrumentation for read path (#2035) --- rfcs/0033-query_tracing.md | 310 +++++++++++++++++++++++++++++++++++++ 1 file changed, 310 insertions(+) create mode 100644 rfcs/0033-query_tracing.md diff --git a/rfcs/0033-query_tracing.md b/rfcs/0033-query_tracing.md new file mode 100644 index 0000000000..5003d7aceb --- /dev/null +++ b/rfcs/0033-query_tracing.md @@ -0,0 +1,310 @@ +# SlateDB Query Tracing + +Table of Contents: + + + +- [Summary](#summary) +- [Motivation](#motivation) +- [Goals](#goals) +- [Non-Goals](#non-goals) +- [Design](#design) + * [Overview](#overview) + * [Query ID in `ReadOptions` and `ScanOptions`](#query-id-in-readoptions-and-scanoptions) + * [Tracing spans](#tracing-spans) +- [Impact Analysis](#impact-analysis) + * [Core API & Query Semantics](#core-api--query-semantics) + * [Consistency, Isolation, and Multi-Versioning](#consistency-isolation-and-multi-versioning) + * [Time, Retention, and Derived State](#time-retention-and-derived-state) + * [Metadata, Coordination, and Lifecycles](#metadata-coordination-and-lifecycles) + * [Compaction](#compaction) + * [Storage Engine Internals](#storage-engine-internals) + * [Ecosystem & Operations](#ecosystem--operations) +- [Operations](#operations) + * [Performance & Cost](#performance--cost) + * [Observability](#observability) + * [Compatibility](#compatibility) +- [Testing](#testing) +- [Rollout](#rollout) +- [Alternatives](#alternatives) + * [Recording aggregations and spans](#recording-aggregations-and-spans) + * [Recording the aggregations in a tracing subscriber](#recording-the-aggregations-in-a-tracing-subscriber) +- [Open Questions](#open-questions) +- [References](#references) +- [Updates](#updates) + + + +Status: Draft + +Authors: + +* [Almog Gavra](https://github.com/agavra) +* [Bruno Cadonna](https://github.com/cadonna) + +## Summary + +This RFC proposes to instrument the read path of SlateDB with the `tracing` library. The read path is instrumented +per query, i.e., per call to a `get` or `scan` method. The instrumentation consists of a root span that +tracks a complete `get` or `scan` operation (including calls on the returned iterator) and child spans that track various +stages of the read path. For example, looking up an entry in the memtable produces a child span. Another example is +reading a filter of an SST. + +The instrumentation is disabled by default. It can be enabled per `get` or `scan` operation by setting the tracing +options in the options of the read operation, i.e., in `ReadOptions` and in `ScanOptions`. + +The generated spans contain fields with information about the span. Each span of a specific read +operation contains the trace ID set in the tracing options passed to the read operation. In addition, the spans contain +information specific to the span, such as the ID of the SST if the span traces processing related to an SST. + +We do not propose any tracing subscriber. The produced spans can be processed by an existing tracing subscriber. +For example, `tracing-chrome` can be used to visualize the spans. + +## Motivation + +SlateDB tracks read-path statistics (bloom filter hits, request counts) +via global `DbStats` counters backed by the `MetricsRecorder` system +(RFC-0021). These aggregate counters answer "how is the system doing?" +but not "why was *this* query slow?" or "how many SSTs did my point +lookup touch?" + +Users today cannot: + +- Determine whether a slow get was caused by bloom filter false + positives, cache misses, or scanning too many L0 SSTs +- Measure how much wall-clock time a scan spent reading blocks from + object storage vs. serving from cache +- Write tests that assert query execution characteristics (e.g. + "this get should hit the bloom filter and skip the SST") + +## Goals + +- Per-query instrumentation via `tracing` spans (e.g., filter evaluations, index reads, + block reads). +- Zero overhead when not opted in (single `Option` branch skip). +- No changes to `DbRead` trait signatures or public API beyond adding + a field to existing options structs. + +## Non-Goals + +- Replacing or duplicating the global `MetricsRecorder` system. Both `DbStats` (aggregate) and the tracing spans + report to distinct consumers independently. +- Write-path tracing (puts, deletes, flush). +- A new `tracing` subscriber/layer. Users should use an existing `tracing` subscriber/layer, such as `tracing-chrome`, + or implement their own subscriber/layer to process and visualize spans. + +## Design + +### Overview + +This proposal adds two concepts to the read path: + +1. Optional tracing options to `ReadOptions` and `ScanOptions`. The default of the tracing options + is `None`, i.e., tracing is disabled by default. +2. `tracing` spans that are conditionally created when the options passed to `get*()` and `scan*()` carry tracing + options that is not `None`. + +### Tracing options in `ReadOptions` and `ScanOptions` + +`ReadOptions` and `ScanOptions` are extended with optional tracing options. +If the tracing options are set, the read path is instrumented. Otherwise, +SlateDB does not create any instrumentation on the read path. + +```rust +pub struct TracingOptions { // new + pub trace_id: String, +} + +pub struct ReadOptions { + pub durability_filter: DurabilityLevel, + pub dirty: bool, + pub cache_blocks: bool, + pub filter_context: Option, + pub tracing_options: Option, // new +} + +pub struct ScanOptions { + pub durability_filter: DurabilityLevel, + pub dirty: bool, + pub read_ahead_bytes: usize, + pub cache_blocks: bool, + pub max_fetch_tasks: usize, + pub order: IterationOrder, + pub filter_context: Option, + pub tracing_options: Option, // new +} +``` + +Both get a `with_tracing_options(TracingOptions) -> Self` builder method. Default is +`None`. + +### Tracing spans + +| Span | Recorded fields | +|--------------------------------|-------------------------------------------------------------| +| `slatedb.read` | `trace_id` | +| `slatedb.read.memtable` | `trace_id` | +| `slatedb.read.read_filters` | `trace_id`, `sst_id`, `level`, `cached` | +| `slatedb.read.evaluate_filter` | `trace_id`, `sst_id`, `level`, `name`, `result` | +| `slatedb.read.read_index` | `trace_id`, `sst_id`, `level`, `cached` | +| `slatedb.read.read_blocks` | `trace_id`, `sst_id`, `level`, `cache_hits`, `cache_misses` | +| `slatedb.read.merge` | `trace_id`, `num_operands` | + +The read path spans are structured hierarchically. The root span for the read path is named `slatedb.read`. +All others are direct children of `slatedb.read`. All spans carry the trace ID +(`trace_id`) as field. The spans are all constructed at debug level. +Spans instrumented on a future are entered each time the future is polled by the runtime. + +The root span `slatedb.read` traces the entire read operation, which includes all stages of the read path +covered by the child spans and common operations over all sources needed for reading, such as setting up +iterators. For scans, the span also covers the read operations triggered by calls on the returned lazy iterator. + +Span `slatedb.read.memtable` traces lookups on the active memtable and the immutable memtables. + +Spans `slatedb.read.read_filters` and `slatedb.read.read_index` trace the reading of filters and reading of the index +of an SST, respectively. The spans carry the ID of the SST (`sst_id`) the filters and the index belong to, the +level on which the SST resides (`level=l0` or `level=sorted_run:{id}`), and field `cached` +that records if the filters or index were found in the cache (`cached=true`) or not (`cached=false`). If the cache +is disabled, `cached` will be `false`. + +The evaluation of a single filter is tracked by span `slatedb.read.evaluate_filter`. The span exposes fields for +the SST ID, the level of the SST, the name of the filter (`name`), and the result of the evaluation (`result`). +For the built-in bloom filter, the `name` field will contain `_bf`. + +Span `slatedb.read.read_blocks` traces the reading of data blocks of an SST. The fields of the span hold the SST ID, +the level of the SST, and how many cache hits and misses were encountered while reading the blocks. + +Processing of the merge operator is traced by span `slatedb.read.merge`. Merging is performed in batches. For each +batch a separate span is produced. Each span contains the number of merged operands as a field. + +## Impact Analysis + +SlateDB features and components that this RFC interacts with. Check all that apply. + +### Core API & Query Semantics + +- [x] Basic KV API (`get`/`put`/`delete`) +- [x] Range queries, iterators, seek semantics +- [ ] Range deletions +- [ ] Error model, API errors + +### Consistency, Isolation, and Multi-Versioning + +- [ ] Transactions +- [ ] Snapshots +- [ ] Sequence numbers + +### Time, Retention, and Derived State + +- [ ] Time to live (TTL) +- [ ] Compaction filters +- [x] Merge operator +- [ ] Change Data Capture (CDC) + +### Metadata, Coordination, and Lifecycles + +- [ ] Manifest format +- [ ] Checkpoints +- [ ] Clones +- [ ] Garbage collection +- [ ] Database splitting and merging +- [ ] Multi-writer + +### Compaction + +- [ ] Compaction state persistence +- [ ] Compaction filters +- [ ] Compaction strategies +- [ ] Distributed compaction +- [ ] Compactions format + +### Storage Engine Internals + +- [ ] Write-ahead log (WAL) +- [x] Block cache +- [x] Object store cache +- [x] Indexing (bloom filters, metadata) +- [ ] SST format or block format + +### Ecosystem & Operations + +- [ ] CLI tools +- [x] Language bindings (Go/Python/etc) +- [x] Observability (metrics/logging/tracing) + +## Operations + +### Performance & Cost + +The proposed instrumentation is disabled by default. Reads without tracing options should not change performance or cost. +When tracing options are set but no tracing subscriber/layer is configured, performance should not be significantly affected. +Set tracing options and a configured tracing subscriber/layer might negatively affect performance. The performance of +writes and compactions should not be affected at all. + +### Observability + +Observability is extended by a per-query instrumentation that traces the read path. The instrumentation only produces +traces at debug level and only if a tracing subscriber/layer is configured. + +### Compatibility + +Field `tracing_options` is added to the public API `ReadOptions` and `ScanOptions`. The field is also exposed in the bindings. +Since the default value of field `tracing_options` is `None`, read path tracing is disabled for existing queries. + +## Testing + +- Unit tests: + - For each kind of span +- Integration tests: + - With subscriber and different queries +- Performance tests: + - With and without tracing options, + - With and without subscriber + - At info and debug level + +## Rollout + +- Milestones / phases: + - Adding root span `slatedb.read`. + - Adding span `slatedb.read.memtable`. + - Adding span `slatedb.read.read_filters`. + - Adding span `slatedb.read.evaluate_filter`. + - Adding span `slatedb.read.read_index`. + - Adding span `slatedb.read.read_blocks`. + - Adding span `slatedb.read.merge`. + - Performance experiments. +- Docs updates: + - Documentation of spans + - Usage example + +## Alternatives + +### Recording aggregations and spans + +The idea was to pass a struct to `ReadOptions` and `ScanOptions` to record and aggregate various measurements, such as +the number of accesses to different sources (e.g., memtables and SSTs), cache misses, and cache hits, as well as instrumenting +spans for recording execution times. This was rejected because collecting aggregations overlapped with instrumenting +the code with spans. We decided to consolidate measurement collection on the read path and also wanted to reduce code +complexity. + +### Recording the aggregations in a tracing subscriber + +This approach consisted of instrumenting the read path and creating a tracing subscriber specifically for SlateDB that +also maintains aggregations. The tracing subscriber would process the instrumented spans and offer an API to read the +recorded data on the read path. This approach was rejected because of the complexity and maintenance burden. +There are already existing tracing subscribers, e.g., `tracing-chrome`, that can be used for analyzing an instrumented +read path. We decided to start with the instrumentation and postpone a dedicated tracing subscriber to the future if +required. + +## Open Questions + +- Should we add a field to the `read_filters` span that records how many filters are read from the SST? + +## References + +- `tracing` crate: https://crates.io/crates/tracing +- `tracing-chrome`: https://crates.io/crates/tracing-chrome +- https://github.com/slatedb/slatedb/issues/400 +- https://github.com/slatedb/slatedb/issues/797 + +## Updates From 31656fe30064ce0d7991a578085341e21438af5e Mon Sep 17 00:00:00 2001 From: Bruno Cadonna Date: Wed, 26 Aug 2026 01:21:34 +0200 Subject: [PATCH 44/65] add tracing options to read path (#2046) --- bindings/go/uniffi/slatedb.go | 92 +++++++++++++++++++ .../java/io/slatedb/uniffi/TestSupport.java | 3 +- bindings/python/tests/conftest.py | 2 + bindings/uniffi/src/config.rs | 24 +++++ bindings/uniffi/src/lib.rs | 2 +- slatedb/src/config.rs | 66 +++++++++++++ slatedb/src/db.rs | 1 + 7 files changed, 188 insertions(+), 2 deletions(-) diff --git a/bindings/go/uniffi/slatedb.go b/bindings/go/uniffi/slatedb.go index 69b251b3c9..0bfad5b4e1 100644 --- a/bindings/go/uniffi/slatedb.go +++ b/bindings/go/uniffi/slatedb.go @@ -10354,6 +10354,8 @@ type ReadOptions struct { // Optional context forwarded to custom filter policies; ignored by // built-in filters. FilterContext *FilterContext + // Optional caller-supplied tracing settings. + TracingOptions *TracingOptions } func (r *ReadOptions) Destroy() { @@ -10361,6 +10363,7 @@ func (r *ReadOptions) Destroy() { FfiDestroyerBool{}.Destroy(r.Dirty) FfiDestroyerBool{}.Destroy(r.CacheBlocks) FfiDestroyerOptionalFilterContext{}.Destroy(r.FilterContext) + FfiDestroyerOptionalTracingOptions{}.Destroy(r.TracingOptions) } type FfiConverterReadOptions struct{} @@ -10377,6 +10380,7 @@ func (c FfiConverterReadOptions) Read(reader io.Reader) ReadOptions { FfiConverterBoolINSTANCE.Read(reader), FfiConverterBoolINSTANCE.Read(reader), FfiConverterOptionalFilterContextINSTANCE.Read(reader), + FfiConverterOptionalTracingOptionsINSTANCE.Read(reader), } } @@ -10393,6 +10397,7 @@ func (c FfiConverterReadOptions) Write(writer io.Writer, value ReadOptions) { FfiConverterBoolINSTANCE.Write(writer, value.Dirty) FfiConverterBoolINSTANCE.Write(writer, value.CacheBlocks) FfiConverterOptionalFilterContextINSTANCE.Write(writer, value.FilterContext) + FfiConverterOptionalTracingOptionsINSTANCE.Write(writer, value.TracingOptions) } type FfiDestroyerReadOptions struct{} @@ -10551,6 +10556,8 @@ type ScanOptions struct { // Optional context forwarded to custom filter policies; ignored by // built-in filters. Only consulted for prefix scans. FilterContext *FilterContext + // Optional caller-supplied tracing settings. + TracingOptions *TracingOptions } func (r *ScanOptions) Destroy() { @@ -10561,6 +10568,7 @@ func (r *ScanOptions) Destroy() { FfiDestroyerUint64{}.Destroy(r.MaxFetchTasks) FfiDestroyerOptionalIterationOrder{}.Destroy(r.Order) FfiDestroyerOptionalFilterContext{}.Destroy(r.FilterContext) + FfiDestroyerOptionalTracingOptions{}.Destroy(r.TracingOptions) } type FfiConverterScanOptions struct{} @@ -10580,6 +10588,7 @@ func (c FfiConverterScanOptions) Read(reader io.Reader) ScanOptions { FfiConverterUint64INSTANCE.Read(reader), FfiConverterOptionalIterationOrderINSTANCE.Read(reader), FfiConverterOptionalFilterContextINSTANCE.Read(reader), + FfiConverterOptionalTracingOptionsINSTANCE.Read(reader), } } @@ -10599,6 +10608,7 @@ func (c FfiConverterScanOptions) Write(writer io.Writer, value ScanOptions) { FfiConverterUint64INSTANCE.Write(writer, value.MaxFetchTasks) FfiConverterOptionalIterationOrderINSTANCE.Write(writer, value.Order) FfiConverterOptionalFilterContextINSTANCE.Write(writer, value.FilterContext) + FfiConverterOptionalTracingOptionsINSTANCE.Write(writer, value.TracingOptions) } type FfiDestroyerScanOptions struct{} @@ -11013,6 +11023,47 @@ func (_ FfiDestroyerSsTableView) Destroy(value SsTableView) { value.Destroy() } +// Options for tracing a read operation. +type TracingOptions struct { + TraceId string +} + +func (r *TracingOptions) Destroy() { + FfiDestroyerString{}.Destroy(r.TraceId) +} + +type FfiConverterTracingOptions struct{} + +var FfiConverterTracingOptionsINSTANCE = FfiConverterTracingOptions{} + +func (c FfiConverterTracingOptions) Lift(rb RustBufferI) TracingOptions { + return LiftFromRustBuffer[TracingOptions](c, rb) +} + +func (c FfiConverterTracingOptions) Read(reader io.Reader) TracingOptions { + return TracingOptions{ + FfiConverterStringINSTANCE.Read(reader), + } +} + +func (c FfiConverterTracingOptions) Lower(value TracingOptions) C.RustBuffer { + return LowerIntoRustBuffer[TracingOptions](c, value) +} + +func (c FfiConverterTracingOptions) LowerExternal(value TracingOptions) ExternalCRustBuffer { + return RustBufferFromC(LowerIntoRustBuffer[TracingOptions](c, value)) +} + +func (c FfiConverterTracingOptions) Write(writer io.Writer, value TracingOptions) { + FfiConverterStringINSTANCE.Write(writer, value.TraceId) +} + +type FfiDestroyerTracingOptions struct{} + +func (_ FfiDestroyerTracingOptions) Destroy(value TracingOptions) { + value.Destroy() +} + // A compactions snapshot paired with its version ID. type VersionedCompactions struct { // Compactions file version ID. @@ -13554,6 +13605,47 @@ func (_ FfiDestroyerOptionalMetric) Destroy(value *Metric) { } } +type FfiConverterOptionalTracingOptions struct{} + +var FfiConverterOptionalTracingOptionsINSTANCE = FfiConverterOptionalTracingOptions{} + +func (c FfiConverterOptionalTracingOptions) Lift(rb RustBufferI) *TracingOptions { + return LiftFromRustBuffer[*TracingOptions](c, rb) +} + +func (_ FfiConverterOptionalTracingOptions) Read(reader io.Reader) *TracingOptions { + if readInt8(reader) == 0 { + return nil + } + temp := FfiConverterTracingOptionsINSTANCE.Read(reader) + return &temp +} + +func (c FfiConverterOptionalTracingOptions) Lower(value *TracingOptions) C.RustBuffer { + return LowerIntoRustBuffer[*TracingOptions](c, value) +} + +func (c FfiConverterOptionalTracingOptions) LowerExternal(value *TracingOptions) ExternalCRustBuffer { + return RustBufferFromC(LowerIntoRustBuffer[*TracingOptions](c, value)) +} + +func (_ FfiConverterOptionalTracingOptions) Write(writer io.Writer, value *TracingOptions) { + if value == nil { + writeInt8(writer, 0) + } else { + writeInt8(writer, 1) + FfiConverterTracingOptionsINSTANCE.Write(writer, *value) + } +} + +type FfiDestroyerOptionalTracingOptions struct{} + +func (_ FfiDestroyerOptionalTracingOptions) Destroy(value *TracingOptions) { + if value != nil { + FfiDestroyerTracingOptions{}.Destroy(*value) + } +} + type FfiConverterOptionalVersionedCompactions struct{} var FfiConverterOptionalVersionedCompactionsINSTANCE = FfiConverterOptionalVersionedCompactions{} diff --git a/bindings/java/slatedb-uniffi/src/test/java/io/slatedb/uniffi/TestSupport.java b/bindings/java/slatedb-uniffi/src/test/java/io/slatedb/uniffi/TestSupport.java index 0534dab438..b0b336794b 100644 --- a/bindings/java/slatedb-uniffi/src/test/java/io/slatedb/uniffi/TestSupport.java +++ b/bindings/java/slatedb-uniffi/src/test/java/io/slatedb/uniffi/TestSupport.java @@ -346,7 +346,7 @@ static String uniquePath(String prefix) { } static ReadOptions readOptions() { - return new ReadOptions(DurabilityLevel.MEMORY, false, true, null); + return new ReadOptions(DurabilityLevel.MEMORY, false, true, null, null); } static ScanOptions scanOptions(long readAheadBytes, boolean cacheBlocks, long maxFetchTasks) { @@ -357,6 +357,7 @@ static ScanOptions scanOptions(long readAheadBytes, boolean cacheBlocks, long ma cacheBlocks, maxFetchTasks, null, + null, null); } diff --git a/bindings/python/tests/conftest.py b/bindings/python/tests/conftest.py index f71d26683b..3e45366503 100644 --- a/bindings/python/tests/conftest.py +++ b/bindings/python/tests/conftest.py @@ -53,6 +53,7 @@ def read_options() -> ReadOptions: durability_filter=DurabilityLevel.MEMORY, dirty=False, cache_blocks=True, + tracing_options=None, ) @@ -65,6 +66,7 @@ def scan_options( read_ahead_bytes=read_ahead_bytes, cache_blocks=cache_blocks, max_fetch_tasks=max_fetch_tasks, + tracing_options=None, ) diff --git a/bindings/uniffi/src/config.rs b/bindings/uniffi/src/config.rs index a6209b5ee0..1ed64e0898 100644 --- a/bindings/uniffi/src/config.rs +++ b/bindings/uniffi/src/config.rs @@ -119,6 +119,20 @@ impl From for slatedb::config::Ttl { } } +/// Options for tracing a read operation. +#[derive(Clone, Debug, uniffi::Record)] +pub struct TracingOptions { + pub trace_id: String, +} + +impl From for slatedb::config::TracingOptions { + fn from(value: TracingOptions) -> Self { + Self { + trace_id: value.trace_id, + } + } +} + /// Options that control a point read. #[derive(Clone, Debug, uniffi::Record)] pub struct ReadOptions { @@ -133,6 +147,9 @@ pub struct ReadOptions { /// built-in filters. #[uniffi(default = None)] pub filter_context: Option, + /// Optional caller-supplied tracing settings. + #[uniffi(default = None)] + pub tracing_options: Option, } impl Default for ReadOptions { @@ -142,6 +159,7 @@ impl Default for ReadOptions { dirty: false, cache_blocks: true, filter_context: None, + tracing_options: None, } } } @@ -153,6 +171,7 @@ impl From for slatedb::config::ReadOptions { dirty: value.dirty, cache_blocks: value.cache_blocks, filter_context: value.filter_context.map(Into::into), + tracing_options: value.tracing_options.map(Into::into), } } } @@ -265,6 +284,9 @@ pub struct ScanOptions { /// built-in filters. Only consulted for prefix scans. #[uniffi(default = None)] pub filter_context: Option, + /// Optional caller-supplied tracing settings. + #[uniffi(default = None)] + pub tracing_options: Option, } impl Default for ScanOptions { @@ -277,6 +299,7 @@ impl Default for ScanOptions { max_fetch_tasks: 1, order: None, filter_context: None, + tracing_options: None, } } } @@ -301,6 +324,7 @@ impl TryFrom for slatedb::config::ScanOptions { })?, order: value.order.unwrap_or_default().into(), filter_context: value.filter_context.map(Into::into), + tracing_options: value.tracing_options.map(Into::into), }) } } diff --git a/bindings/uniffi/src/lib.rs b/bindings/uniffi/src/lib.rs index 59d00be170..c746fa147c 100644 --- a/bindings/uniffi/src/lib.rs +++ b/bindings/uniffi/src/lib.rs @@ -27,7 +27,7 @@ pub use config::{ CloseOptions, DurabilityLevel, FlushOptions, FlushType, GarbageCollectorDirectoryOptions, GarbageCollectorOptions, GarbageCollectorScheduleOptions, IsolationLevel, IterationOrder, MergeOptions, PutOptions, ReadOptions, ReaderMode, ReaderOptions, ScanOptions, SstBlockSize, - Ttl, WriteOptions, + TracingOptions, Ttl, WriteOptions, }; pub use db::Db; pub use db_reader::DbReader; diff --git a/slatedb/src/config.rs b/slatedb/src/config.rs index c595134096..bfe418d9cb 100644 --- a/slatedb/src/config.rs +++ b/slatedb/src/config.rs @@ -281,6 +281,20 @@ pub enum DurabilityLevel { Memory, } +/// Options for tracing a read operation. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct TracingOptions { + pub trace_id: String, +} + +impl TracingOptions { + pub fn new(trace_id: impl Into) -> Self { + Self { + trace_id: trace_id.into(), + } + } +} + /// Configuration for client read operations. `ReadOptions` is supplied for each /// read call and controls the behavior of the read. #[derive(Clone, Debug)] @@ -298,6 +312,8 @@ pub struct ReadOptions { /// Optional context forwarded to custom filter policies; ignored by /// built-in filters. See [`FilterContext`]. pub filter_context: Option, + /// Optional caller-provided tracing settings. + pub tracing_options: Option, } impl Default for ReadOptions { @@ -307,6 +323,7 @@ impl Default for ReadOptions { dirty: false, cache_blocks: true, filter_context: None, + tracing_options: None, } } } @@ -340,6 +357,13 @@ impl ReadOptions { ..self } } + + pub fn with_tracing_options(self, tracing_options: Option) -> Self { + Self { + tracing_options, + ..self + } + } } #[derive(Clone, Debug)] pub struct ScanOptions { @@ -369,6 +393,8 @@ pub struct ScanOptions { /// Only consulted for `scan_prefix` today. Plain range scans do not /// evaluate SST filters, so this field has no effect on `scan`. pub filter_context: Option, + /// Optional caller-provided tracing settings. + pub tracing_options: Option, } impl Default for ScanOptions { @@ -382,6 +408,7 @@ impl Default for ScanOptions { max_fetch_tasks: 1, order: IterationOrder::Ascending, filter_context: None, + tracing_options: None, } } } @@ -433,6 +460,13 @@ impl ScanOptions { ..self } } + + pub fn with_tracing_options(self, tracing_options: Option) -> Self { + Self { + tracing_options, + ..self + } + } } /// Enum representing the type of flush to perform. @@ -1994,6 +2028,12 @@ object_store_cache_options: assert_eq!(options.max_fetch_tasks, 1); } + #[test] + fn test_default_tracing_options_are_none() { + assert!(ReadOptions::default().tracing_options.is_none()); + assert!(ScanOptions::default().tracing_options.is_none()); + } + #[test] fn test_scan_options_with_max_fetch_tasks() { let options = ScanOptions::default().with_max_fetch_tasks(4); @@ -2006,6 +2046,32 @@ object_store_cache_options: assert!(!options.cache_blocks); } + #[test] + fn test_read_options_with_tracing_options() { + let options = + ReadOptions::default().with_tracing_options(Some(TracingOptions::new("read-trace"))); + assert_eq!( + options + .tracing_options + .as_ref() + .map(|tracing_options| tracing_options.trace_id.as_str()), + Some("read-trace") + ); + } + + #[test] + fn test_scan_options_with_tracing_options() { + let options = + ScanOptions::default().with_tracing_options(Some(TracingOptions::new("scan-trace"))); + assert_eq!( + options + .tracing_options + .as_ref() + .map(|tracing_options| tracing_options.trace_id.as_str()), + Some("scan-trace") + ); + } + #[test] fn test_size_tiered_compaction_scheduler_options_roundtrip() { let options = SizeTieredCompactionSchedulerOptions { diff --git a/slatedb/src/db.rs b/slatedb/src/db.rs index f1485b1ab6..37264d921d 100644 --- a/slatedb/src/db.rs +++ b/slatedb/src/db.rs @@ -3155,6 +3155,7 @@ mod tests { dirty: false, cache_blocks: true, filter_context: None, + tracing_options: None, } ) .await From ae07acd4498068d1b9ba799cc9f6c9824e6f6251 Mon Sep 17 00:00:00 2001 From: Chris Date: Wed, 26 Aug 2026 14:15:22 -0700 Subject: [PATCH 45/65] fix: preserve sequence tracker interval after deserialization (#2053) --- slatedb/src/seq_tracker.rs | 18 ++++++++++++++++++ 1 file changed, 18 insertions(+) diff --git a/slatedb/src/seq_tracker.rs b/slatedb/src/seq_tracker.rs index 2c70ba46d4..b123c9729a 100644 --- a/slatedb/src/seq_tracker.rs +++ b/slatedb/src/seq_tracker.rs @@ -289,6 +289,7 @@ fn decode_sequence_tracker(buf: &[u8]) -> Result { let mut tracker = SequenceTracker::with_config(DEFAULT_CAPACITY, DEFAULT_INTERVAL_SECS); tracker.sequence_numbers = sequence_numbers; tracker.timestamps = timestamps; + tracker.last_recorded_ts = tracker.timestamps.last().copied(); Ok(tracker) } @@ -779,6 +780,23 @@ mod tests { assert_eq!(decoded.timestamps, tracker.timestamps); } + #[test] + fn deserialize_preserves_the_recording_interval() { + let mut tracker = SequenceTracker::new(); + tracker.insert(TrackedSeq { + seq: 0, + ts: DateTime::from_timestamp(1_600_000_060, 0).unwrap(), + }); + + let mut decoded = SequenceTracker::from_bytes(&tracker.to_bytes()).unwrap(); + decoded.insert(TrackedSeq { + seq: 1, + ts: DateTime::from_timestamp(1_600_000_061, 0).unwrap(), + }); + + assert_eq!(decoded.sequence_numbers, vec![0]); + } + #[rstest] #[case::empty_sequences(vec![])] #[case::single_sequence(vec![1000])] From c325738b41d0cd9dfe56cd3876a5dac6fb4a1d02 Mon Sep 17 00:00:00 2001 From: Roman Date: Thu, 27 Aug 2026 20:43:24 +0200 Subject: [PATCH 46/65] manifest: write ManifestV2 universally (#2055) (#2056) --- bindings/go/uniffi/slatedb_test.go | 5 ++-- slatedb/src/db.rs | 20 +++++++------ slatedb/src/flatbuffer_types.rs | 46 +++++++++++++++++------------- 3 files changed, 40 insertions(+), 31 deletions(-) diff --git a/bindings/go/uniffi/slatedb_test.go b/bindings/go/uniffi/slatedb_test.go index 3bf89a4d4c..b05ad1d267 100644 --- a/bindings/go/uniffi/slatedb_test.go +++ b/bindings/go/uniffi/slatedb_test.go @@ -1880,8 +1880,9 @@ func TestAdminQueries(t *testing.T) { if latestManifest.LastL0Seq < 3 { t.Fatalf("ReadManifest(nil): LastL0Seq = %d, want at least 3", latestManifest.LastL0Seq) } - if latestManifest.WalObjectStoreUri == nil { - t.Fatal("ReadManifest(nil): WalObjectStoreUri = nil, want value for configured WAL store") + // ManifestV2 is now written universally and never persists wal_object_store_uri + if latestManifest.WalObjectStoreUri != nil { + t.Fatalf("ReadManifest(nil): WalObjectStoreUri = %v, want nil (dropped by ManifestV2)", *latestManifest.WalObjectStoreUri) } firstManifest, err := admin.ReadManifest(uint64Ptr(manifests[0].Id)) diff --git a/slatedb/src/db.rs b/slatedb/src/db.rs index 37264d921d..57f73083cc 100644 --- a/slatedb/src/db.rs +++ b/slatedb/src/db.rs @@ -7347,8 +7347,14 @@ mod tests { kv_store.close().await.unwrap(); } + /// https://github.com/slatedb/slatedb/issues/2055: manifests are now + /// written as V2 universally (RFC-0004 Phase 2), which drops + /// `wal_object_store_uri` entirely (PR #1473). So a DB created with + /// `with_wal_object_store()` can be reopened without it configured; + /// the WAL-store reconfiguration check is a no-op once the persisted + /// manifest is V2. #[tokio::test] - async fn test_wal_store_reconfiguration_fails() { + async fn test_wal_store_reconfiguration_allowed_after_v2_manifest() { let object_store = Arc::new(InMemory::new()); let wal_object_store = Arc::new(InMemory::new()); @@ -7360,16 +7366,12 @@ mod tests { .unwrap(); kv_store.close().await.unwrap(); - let result = Db::builder("/tmp/test_kv_store", object_store) + let reopened = Db::builder("/tmp/test_kv_store", object_store) .with_settings(test_db_options(0, 1024, None)) .build() - .await; - match result { - Err(err) => { - assert!(err.to_string().contains("unsupported")); - } - _ => panic!("expected Unsupported error"), - } + .await + .unwrap(); + reopened.close().await.unwrap(); } #[tokio::test] diff --git a/slatedb/src/flatbuffer_types.rs b/slatedb/src/flatbuffer_types.rs index 1bf6bd2023..812d520f43 100644 --- a/slatedb/src/flatbuffer_types.rs +++ b/slatedb/src/flatbuffer_types.rs @@ -42,13 +42,17 @@ use crate::flatbuffer_types::root_generated::{ CompactedSsTableViewArgs, Compaction as FbCompaction, CompactionArgs as FbCompactionArgs, CompactionContext as FbCompactionContext, CompactionSpec as FbCompactionSpec, CompactionStatus as FbCompactionStatus, CompactionsV1, CompactionsV1Args, CompressionFormat, - DrainSegmentSpec, DrainSegmentSpecArgs, ManifestV1Args, Segment as FbSegment, - SegmentArgs as FbSegmentArgs, SortedRun as FbSortedRunV1, SortedRunArgs as FbSortedRunV1Args, + DrainSegmentSpec, DrainSegmentSpecArgs, Segment as FbSegment, SegmentArgs as FbSegmentArgs, SortedRunV2, SortedRunV2Args, SstType as FbSstType, Subcompaction as FbSubcompaction, SubcompactionArgs as FbSubcompactionArgs, TieredCompactionContext as FbTieredCompactionContext, TieredCompactionContextArgs as FbTieredCompactionContextArgs, TieredCompactionSpec, TieredCompactionSpecArgs, Ulid as FbUlid, UlidArgs as FbUlidArgs, Uuid, UuidArgs, }; +// V1-only manifest encoder types; only test fixtures build V1 manifests +#[cfg(test)] +use crate::flatbuffer_types::root_generated::{ + ManifestV1Args, SortedRun as FbSortedRunV1, SortedRunArgs as FbSortedRunV1Args, +}; use crate::format::sst::SST_FORMAT_VERSION; use crate::manifest::{ExternalDb, LsmTreeState, Manifest, ManifestCore, Segment}; use crate::partitioned_keyspace::RangePartitionedKeySpace; @@ -170,16 +174,10 @@ pub(crate) struct FlatBufferManifestCodec {} impl ObjectCodec for FlatBufferManifestCodec { fn encode(&self, manifest: &Manifest) -> Bytes { - // RFC-0024 lazy V2 bump: the V1 schema has no `segments` or - // `segment_extractor_name` fields, so writing segmented state - // through the V1 encoder would silently drop it. Pick V2 the - // moment any segmented state is present; databases that never - // configure an extractor keep writing V1. - if Self::requires_v2(manifest) { - Self::create_from_manifest(manifest) - } else { - Self::create_from_manifest_v1(manifest) - } + // RFC-0004 Phase 2 of the manifest V1->V2 rollout: write V2 + // universally, read V1+V2. V1 read support (and the V1 encoder, + // retained below for tests) stays until no V1 manifests remain + Self::create_from_manifest(manifest) } fn decode(&self, bytes: &Bytes) -> Result> { @@ -523,6 +521,10 @@ impl FlatBufferManifestCodec { }) } + /// Retained for tests that construct V1 manifest fixtures to verify + /// decode-side backward compatibility; production code always writes + /// V2 (see [`FlatBufferManifestCodec::encode`]). + #[cfg(test)] pub(crate) fn create_from_manifest_v1(manifest: &Manifest) -> Bytes { let builder = FlatBufferBuilder::new(); let mut db_fb_builder = DbFlatBufferBuilder::new(builder); @@ -534,12 +536,6 @@ impl FlatBufferManifestCodec { let mut db_fb_builder = DbFlatBufferBuilder::new(builder); db_fb_builder.create_manifest(manifest) } - - /// Whether `manifest` carries state that V1 cannot represent: - /// a configured segment extractor, or any named segment. - fn requires_v2(manifest: &Manifest) -> bool { - manifest.core.segment_extractor_name.is_some() || !manifest.core.segments.is_empty() - } } pub(crate) struct FlatBufferCompactionsCodec {} @@ -864,6 +860,9 @@ impl<'b> DbFlatBufferBuilder<'b> { self.builder.create_vector(compacted_ssts.as_ref()) } + /// V1-only; only test fixtures construct V1 manifests now that + /// `encode()` writes V2 universally (see [`FlatBufferManifestCodec::encode`]). + #[cfg(test)] fn add_compacted_sst_from_view( &mut self, view: &SsTableView, @@ -1011,6 +1010,7 @@ impl<'b> DbFlatBufferBuilder<'b> { self.builder.create_vector(segment_offsets.as_ref()) } + #[cfg(test)] fn add_sorted_run_v1( &mut self, sorted_run: &db_state::SortedRun, @@ -1030,6 +1030,7 @@ impl<'b> DbFlatBufferBuilder<'b> { ) } + #[cfg(test)] fn add_sorted_runs_v1( &mut self, sorted_runs: &[db_state::SortedRun], @@ -1361,6 +1362,7 @@ impl<'b> DbFlatBufferBuilder<'b> { bytes.into() } + #[cfg(test)] fn create_manifest_v1(&mut self, manifest: &Manifest) -> Bytes { let core = &manifest.core; @@ -1822,10 +1824,14 @@ mod tests { let v1_bytes = bytes.freeze(); codec.decode(&v1_bytes).expect("Should decode V1 manifest"); - // Test encode/decode round-trip (currently writes V1 for forward compatibility) + // Test encode/decode round-trip (writes V2 universally, per RFC-0004 + // Phase 2 of the manifest V1->V2 rollout) let manifest = Manifest::initial(ManifestCore::new()); let encoded = codec.encode(&manifest); - assert_eq!(u16::from_be_bytes([encoded[0], encoded[1]]), 1); + assert_eq!( + u16::from_be_bytes([encoded[0], encoded[1]]), + MANIFEST_FORMAT_VERSION + ); codec .decode(&encoded) .expect("Should decode manifest round-trip"); From ea75d1254b02dede00ed78ab4005155b5282e2fa Mon Sep 17 00:00:00 2001 From: Rohan Date: Sat, 29 Aug 2026 00:37:07 -0400 Subject: [PATCH 47/65] move wal read/write paths to their own types under wal module (#2052) --- slatedb/src/db/builder.rs | 57 ++- slatedb/src/db_reader.rs | 85 +++-- slatedb/src/fence.rs | 27 +- slatedb/src/lib.rs | 1 + slatedb/src/sst_io.rs | 111 ++++++ slatedb/src/tablestore.rs | 187 +++------- slatedb/src/wal/slatedb/iterator.rs | 163 ++++----- slatedb/src/wal/slatedb/mod.rs | 2 + slatedb/src/wal/slatedb/reader.rs | 105 +++--- slatedb/src/wal/slatedb/sst_iterator.rs | 171 +++++++++ slatedb/src/wal/slatedb/store.rs | 447 ++++++++++++++++++++++++ slatedb/src/wal/slatedb/writer.rs | 72 ++-- slatedb/src/wal/slatedb/writer_init.rs | 7 +- slatedb/src/wal_replay.rs | 120 ++++--- 14 files changed, 1124 insertions(+), 431 deletions(-) create mode 100644 slatedb/src/sst_io.rs create mode 100644 slatedb/src/wal/slatedb/sst_iterator.rs create mode 100644 slatedb/src/wal/slatedb/store.rs diff --git a/slatedb/src/db/builder.rs b/slatedb/src/db/builder.rs index c59c82ce99..4eecf2ffc3 100644 --- a/slatedb/src/db/builder.rs +++ b/slatedb/src/db/builder.rs @@ -163,6 +163,7 @@ use crate::utils::SafeSender; use crate::utils::WatchableOnceCell; use crate::wal; use crate::wal::slatedb::admin::SlateDbWalAdmin; +use crate::wal::slatedb::store::WalTableStore; use crate::wal::wal_disabled::DisabledWalObserver; use crate::wal::{WalAdmin, WalGc, WalObserver}; use slatedb_common::clock::DefaultSystemClock; @@ -496,7 +497,11 @@ impl> DbBuilder

{ ); let retrying_wal_object_store: Option> = self .wal_object_store + .clone() .map(|s| wrap_object_store(s, ObjectStoreComponent::Db, ObjectStoreType::Wal)); + let wal_object_store = retrying_wal_object_store + .clone() + .unwrap_or_else(|| retrying_main_object_store.clone()); // Log the database opening if let Ok(settings_json) = self.settings.to_json_string() { @@ -588,6 +593,13 @@ impl> DbBuilder

{ TableStoreKind::Main, self.block_cache_policy.clone(), )); + let wal_store = Arc::new(WalTableStore::new_with_fp_registry( + wal_object_store, + sst_format.clone(), + path_resolver.clone(), + self.fp_registry.clone(), + TableStoreKind::Main, + )); // Initialize the database let stored_manifest = match latest_manifest { @@ -622,7 +634,7 @@ impl> DbBuilder

{ let fencer = WriterFencer::new( status_manager.result_reader(), recorder.clone(), - table_store.clone(), + wal_store.clone(), &self.settings, system_clock.clone(), task_executor.clone(), @@ -1838,7 +1850,7 @@ impl> DbReaderBuilder

{ .await?; let maybe_cached_object_store: Arc = match &maybe_cached { Some(cached) => Arc::clone(cached) as Arc, - None => self.object_store, + None => self.object_store.clone(), }; let retrying_object_store = instrumented_retrying_object_store( @@ -1851,18 +1863,20 @@ impl> DbReaderBuilder

{ self.options.object_store_max_retries, ); - let retrying_wal_object_store: Option> = - self.wal_object_store.map(|s| { - instrumented_retrying_object_store( - s, - &recorder, - ObjectStoreComponent::Reader, - ObjectStoreType::Wal, - self.rand.clone(), - self.system_clock.clone(), - self.options.object_store_max_retries, - ) - }); + let retrying_wal_object_store = self.wal_object_store.map(|wal_object_store| { + instrumented_retrying_object_store( + wal_object_store, + &recorder, + ObjectStoreComponent::Reader, + ObjectStoreType::Wal, + self.rand.clone(), + self.system_clock.clone(), + self.options.object_store_max_retries, + ) + }); + let wal_object_store = retrying_wal_object_store + .clone() + .unwrap_or_else(|| retrying_object_store.clone()); // Validate WAL object store configuration. let manifest_store = Arc::new(ManifestStore::new(&path, retrying_object_store.clone())); @@ -1908,19 +1922,28 @@ impl> DbReaderBuilder

{ ..SsTableFormat::default() }; let path_resolver = PathResolver::new_with_external_ssts(path.clone(), external_ssts); + let fp_registry = Arc::new(FailPointRegistry::new()); let table_store = Arc::new(TableStore::new_with_fp_registry( ObjectStores::new(retrying_object_store, retrying_wal_object_store), - sst_format, - path_resolver, - Arc::new(FailPointRegistry::new()), + sst_format.clone(), + path_resolver.clone(), + Arc::clone(&fp_registry), wrapped_cache, TableStoreKind::Reader, BlockCachePolicy::default(), )); + let wal_store = Arc::new(WalTableStore::new_with_fp_registry( + wal_object_store, + sst_format, + path_resolver, + fp_registry, + TableStoreKind::Reader, + )); let reader = DbReader::open_internal( manifest_store, table_store, + wal_store, self.mode, self.wal_reader, self.merge_operator, diff --git a/slatedb/src/db_reader.rs b/slatedb/src/db_reader.rs index 2716e44ffd..092556552d 100644 --- a/slatedb/src/db_reader.rs +++ b/slatedb/src/db_reader.rs @@ -26,6 +26,7 @@ use { tablestore::TableStore, types::KeyValue, utils::IdGenerator, + wal::slatedb::store::WalTableStore, wal::WalReader as WalReaderTrait, wal_replay::{WalReplayIterator, WalReplayOptions}, Checkpoint, DbCacheManagerOps, DbIterator, DbMetadataOps, DbReadOps, @@ -173,6 +174,7 @@ impl DbReaderInner { async fn new( manifest_store: Arc, table_store: Arc, + wal_store: Arc, wal_reader: Option>, options: DbReaderOptions, mode: DbReaderMode, @@ -201,7 +203,7 @@ impl DbReaderInner { let wal_reader = wal_reader.unwrap_or_else(|| { Arc::new( crate::wal::slatedb::reader::SlateDbWalReader::new_with_status_manager( - Arc::clone(&table_store), + wal_store, &status_manager, Arc::clone(&system_clock), SlateDbWalReaderOptions { @@ -965,6 +967,7 @@ impl DbReader { pub(crate) async fn open_internal( manifest_store: Arc, table_store: Arc, + wal_store: Arc, mode: DbReaderMode, wal_reader: Option>, merge_operator: Option, @@ -990,6 +993,7 @@ impl DbReader { DbReaderInner::new( manifest_store, table_store, + wal_store, wal_reader, options, mode, @@ -1486,7 +1490,7 @@ mod tests { MergeOptions, PutOptions, Settings, WriteOptions, }, db_reader::{DbReader, DbReaderInner, DbReaderMode, DbReaderOptions}, - db_state::{SsTableId, SstType}, + db_state::SstType, db_stats::DbStats, db_status::DbStatusManager, dispatcher::MessageHandler, @@ -1507,7 +1511,10 @@ mod tests { tablestore::{TableStore, TableStoreKind}, test_utils, types::RowEntry, - wal::{WalError, WalFileRange, WalIterator, WalReader as WalReaderTrait, WalRows}, + wal::{ + slatedb::store::WalTableStore, WalError, WalFileRange, WalIterator, + WalReader as WalReaderTrait, WalRows, + }, CloseReason, Db, }, bytes::Bytes, @@ -1666,6 +1673,7 @@ mod tests { let reader = DbReader::open_internal( test_provider.manifest_store(), test_provider.table_store(), + test_provider.wal_store(), DbReaderMode::Checkpoint(checkpoint_result.id), None, None, @@ -2026,6 +2034,7 @@ mod tests { let reader = DbReader::open_internal( test_provider.manifest_store(), test_provider.table_store(), + test_provider.wal_store(), DbReaderMode::FollowLatest, None, None, @@ -2223,6 +2232,7 @@ mod tests { let inner = DbReaderInner::new( Arc::clone(&manifest_store), table_store, + test_provider.wal_store(), None, DbReaderOptions { manifest_poll_interval: Duration::from_millis(100), @@ -2319,6 +2329,7 @@ mod tests { let inner = DbReaderInner::new( Arc::clone(&manifest_store), table_store, + test_provider.wal_store(), None, DbReaderOptions { manifest_poll_interval: Duration::from_millis(100), @@ -2405,16 +2416,17 @@ mod tests { let path = Path::from("/tmp/test_db_reader_replay_order"); let test_provider = TestProvider::new(path, Arc::clone(&object_store)); let table_store = test_provider.table_store(); + let wal_store = test_provider.wal_store(); write_wal_sst( - Arc::clone(&table_store), + Arc::clone(&wal_store), 3, vec![RowEntry::new_value(b"stale_key", b"stale_value", 3)], ) .await .unwrap(); write_wal_sst( - Arc::clone(&table_store), + Arc::clone(&wal_store), 4, vec![RowEntry::new_value(b"fresh_key", b"fresh_value", 4)], ) @@ -2437,7 +2449,7 @@ mod tests { let (last_wal_id, last_committed_seq) = DbReaderInner::replay_wal_into( Arc::clone(&table_store), - &native_wal_reader(&table_store, &status_manager), + &native_wal_reader(&wal_store, &status_manager), &DbReaderOptions::default(), &core, &mut into_tables, @@ -2468,9 +2480,10 @@ mod tests { let path = Path::from("/tmp/test_db_reader_missing_wal"); let test_provider = TestProvider::new(path, Arc::clone(&object_store)); let table_store = test_provider.table_store(); + let wal_store = test_provider.wal_store(); write_wal_sst( - Arc::clone(&table_store), + Arc::clone(&wal_store), 1, vec![RowEntry::new_value(b"key", b"value", 1)], ) @@ -2484,7 +2497,7 @@ mod tests { let (last_wal_id, last_committed_seq) = DbReaderInner::replay_wal_into( Arc::clone(&table_store), - &native_wal_reader(&table_store, &status_manager), + &native_wal_reader(&wal_store, &status_manager), &DbReaderOptions::default(), &core, &mut into_tables, @@ -2508,6 +2521,7 @@ mod tests { let path = Path::from("/tmp/test_db_reader_missing_wal_after_replay"); let test_provider = TestProvider::new(path, Arc::clone(&object_store)); let table_store = test_provider.table_store(); + let wal_store = test_provider.wal_store(); let wal_1_row = RowEntry::new_value(b"a", &[b'a'; 8], 1); let wal_2_row_1 = RowEntry::new_value(b"b", &[b'b'; 40], 2); @@ -2518,11 +2532,11 @@ mod tests { wal_1_row.estimated_size() + wal_2_row_1.estimated_size(), ) as u64; - write_wal_sst(Arc::clone(&table_store), 1, vec![wal_1_row.clone()]) + write_wal_sst(Arc::clone(&wal_store), 1, vec![wal_1_row.clone()]) .await .unwrap(); write_wal_sst( - Arc::clone(&table_store), + Arc::clone(&wal_store), 2, vec![wal_2_row_1.clone(), wal_2_row_2.clone()], ) @@ -2541,7 +2555,7 @@ mod tests { let (last_wal_id, last_committed_seq) = DbReaderInner::replay_wal_into( Arc::clone(&table_store), - &native_wal_reader(&table_store, &status_manager), + &native_wal_reader(&wal_store, &status_manager), &reader_options, &core, &mut into_tables, @@ -2572,6 +2586,7 @@ mod tests { let path = Path::from("/tmp/test_db_reader_fresh_db_no_writes"); let test_provider = TestProvider::new(path, Arc::clone(&object_store)); let table_store = test_provider.table_store(); + let wal_store = test_provider.wal_store(); let mut into_tables = VecDeque::new(); let core = ManifestCore::new(); @@ -2579,7 +2594,7 @@ mod tests { let (last_wal_id, last_committed_seq) = DbReaderInner::replay_wal_into( Arc::clone(&table_store), - &native_wal_reader(&table_store, &status_manager), + &native_wal_reader(&wal_store, &status_manager), &DbReaderOptions::default(), &core, &mut into_tables, @@ -2600,9 +2615,10 @@ mod tests { let path = Path::from("/tmp/test_db_reader_fresh_db_one_wal"); let test_provider = TestProvider::new(path, Arc::clone(&object_store)); let table_store = test_provider.table_store(); + let wal_store = test_provider.wal_store(); let wal_row = RowEntry::new_value(b"key", b"value", 1); - write_wal_sst(Arc::clone(&table_store), 1, vec![wal_row.clone()]) + write_wal_sst(Arc::clone(&wal_store), 1, vec![wal_row.clone()]) .await .unwrap(); @@ -2612,7 +2628,7 @@ mod tests { let (last_wal_id, last_committed_seq) = DbReaderInner::replay_wal_into( Arc::clone(&table_store), - &native_wal_reader(&table_store, &status_manager), + &native_wal_reader(&wal_store, &status_manager), &DbReaderOptions::default(), &core, &mut into_tables, @@ -2639,8 +2655,9 @@ mod tests { let path = Path::from("/tmp/test_db_reader_empty_fence_wal"); let test_provider = TestProvider::new(path, Arc::clone(&object_store)); let table_store = test_provider.table_store(); + let wal_store = test_provider.wal_store(); - write_wal_sst(Arc::clone(&table_store), 6, vec![]) + write_wal_sst(Arc::clone(&wal_store), 6, vec![]) .await .unwrap(); @@ -2660,7 +2677,7 @@ mod tests { let (last_wal_id, last_committed_seq) = DbReaderInner::replay_wal_into( Arc::clone(&table_store), - &native_wal_reader(&table_store, &status_manager), + &native_wal_reader(&wal_store, &status_manager), &DbReaderOptions::default(), &core, &mut into_tables, @@ -2990,7 +3007,7 @@ mod tests { let object_store: Arc = recording_store.clone(); let path = Path::from("/tmp/test_kv_store"); let test_provider = TestProvider::new(path.clone(), Arc::clone(&object_store)); - let table_store = test_provider.table_store(); + let wal_store = test_provider.wal_store(); // One 16MiB WAL SST, far over the old 1MiB window. The old window read it in // ~16 data GETs; the fix reads it in one. @@ -2998,14 +3015,14 @@ mod tests { let entries: Vec = (0..4096u32) .map(|i| RowEntry::new_value(format!("key-{i:08}").as_bytes(), &value, i as u64 + 1)) .collect(); - write_wal_sst(Arc::clone(&table_store), 1, entries) + write_wal_sst(Arc::clone(&wal_store), 1, entries) .await .unwrap(); let mut core = ManifestCore::new(); core.next_wal_sst_id = 2; let status_manager = status_manager_for_core(&core); - let wal_reader = native_wal_reader(&table_store, &status_manager); + let wal_reader = native_wal_reader(&wal_store, &status_manager); recording_store.clear(); let mut iterator = wal_reader.iterator((1..2).into()).await.unwrap(); @@ -3139,6 +3156,7 @@ mod tests { DbReader::open_internal( self.manifest_store(), self.table_store(), + self.wal_store(), mode, None, merge_operator, @@ -3161,11 +3179,11 @@ mod tests { } fn native_wal_reader( - table_store: &Arc, + wal_store: &Arc, status_manager: &DbStatusManager, ) -> crate::wal::slatedb::reader::SlateDbWalReader { crate::wal::slatedb::reader::SlateDbWalReader::new_with_status_manager( - Arc::clone(table_store), + Arc::clone(wal_store), status_manager, Arc::new(DefaultSystemClock::new()), SlateDbWalReaderOptions::default(), @@ -3184,15 +3202,16 @@ mod tests { } async fn write_wal_sst( - table_store: Arc, + wal_store: Arc, wal_id: u64, entries: Vec, ) -> Result<(), SlateDBError> { - let mut writer = table_store.table_writer(SsTableId::Wal(wal_id)); + let mut builder = wal_store.table_builder(); for entry in entries { - writer.add(entry).await?; + builder.add(entry).await?; } - writer.close().await?; + let encoded_sst = builder.build().await?; + wal_store.write_sst(wal_id, &encoded_sst).await?; Ok(()) } @@ -3294,6 +3313,7 @@ mod tests { let test_provider = TestProvider::new(path, Arc::clone(&object_store)); let manifest_store = test_provider.manifest_store(); let table_store = test_provider.table_store(); + let wal_store = test_provider.wal_store(); let mut stored_manifest = StoredManifest::create_new_db( Arc::clone(&manifest_store), ManifestCore::new(), @@ -3347,7 +3367,7 @@ mod tests { None, ); let status_manager = status_manager_for_core(&stored_manifest.manifest().core); - let wal_reader = Arc::new(native_wal_reader(&table_store, &status_manager)); + let wal_reader = Arc::new(native_wal_reader(&wal_store, &status_manager)); let inner = DbReaderInner { manifest_store, table_store, @@ -3410,6 +3430,7 @@ mod tests { ) -> DbReaderInner { let manifest_store = test_provider.manifest_store(); let table_store = test_provider.table_store(); + let wal_store = test_provider.wal_store(); let status_manager = status_manager_for_core(current_core); let prior_state = ReaderState { @@ -3436,7 +3457,7 @@ mod tests { oracle.clone(), None, ); - let wal_reader = Arc::new(native_wal_reader(&table_store, &status_manager)); + let wal_reader = Arc::new(native_wal_reader(&wal_store, &status_manager)); DbReaderInner { manifest_store, table_store, @@ -3681,6 +3702,16 @@ mod tests { )) } + fn wal_store(&self) -> Arc { + Arc::new(WalTableStore::new_with_fp_registry( + Arc::clone(&self.object_store), + SsTableFormat::default(), + PathResolver::from_root(self.path.clone()), + Arc::clone(&self.fp_registry), + TableStoreKind::Reader, + )) + } + fn manifest_store(&self) -> Arc { Arc::new(ManifestStore::new( &self.path, diff --git a/slatedb/src/fence.rs b/slatedb/src/fence.rs index 5e558061df..7ee672d2aa 100644 --- a/slatedb/src/fence.rs +++ b/slatedb/src/fence.rs @@ -1,8 +1,8 @@ use crate::dispatcher::MessageHandlerExecutor; use crate::error::SlateDBError; use crate::manifest::store::{FenceableManifest, StoredManifest}; -use crate::tablestore::TableStore; use crate::utils::WatchableOnceCellReader; +use crate::wal::slatedb::store::WalTableStore; use crate::wal::slatedb::writer_init::{SlateDbWalWriterInit, SlateDbWalWriterInitOptions}; use crate::wal::{WalIterator, WalWriter, WriterInit}; use crate::Settings; @@ -16,7 +16,7 @@ pub(crate) struct WriterFencer { closed_result_reader: WatchableOnceCellReader>, recorder: MetricsRecorderHelper, wal_writer_init_options: SlateDbWalWriterInitOptions, - table_store: Arc, + wal_store: Arc, manifest_update_timeout: Duration, system_clock: Arc, task_executor: Arc, @@ -35,7 +35,7 @@ impl WriterFencer { pub(crate) fn new( closed_result_reader: WatchableOnceCellReader>, recorder: MetricsRecorderHelper, - table_store: Arc, + wal_store: Arc, settings: &Settings, system_clock: Arc, task_executor: Arc, @@ -44,7 +44,7 @@ impl WriterFencer { Self::new_with_fp_handle( closed_result_reader, recorder, - table_store, + wal_store, settings, system_clock, task_executor, @@ -56,7 +56,7 @@ impl WriterFencer { fn new_with_fp_handle( closed_result_reader: WatchableOnceCellReader>, recorder: MetricsRecorderHelper, - table_store: Arc, + wal_store: Arc, settings: &Settings, system_clock: Arc, task_executor: Arc, @@ -66,7 +66,7 @@ impl WriterFencer { Self { closed_result_reader, recorder, - table_store, + wal_store, wal_writer_init_options: settings.into(), manifest_update_timeout: settings.manifest_update_timeout, system_clock, @@ -94,7 +94,7 @@ impl WriterFencer { SlateDbWalWriterInit::load( self.closed_result_reader.clone(), self.recorder.clone(), - self.table_store.clone(), + self.wal_store.clone(), self.wal_writer_init_options, stored_manifest.manifest(), self.task_executor.clone(), @@ -144,8 +144,10 @@ mod tests { use crate::manifest::ManifestCore; use crate::memtable_flusher::MANIFEST_REFRESH_COUNT; use crate::object_stores::ObjectStores; + use crate::paths::PathResolver; use crate::tablestore::{TableStore, TableStoreKind}; use crate::utils::WatchableOnceCell; + use crate::wal::slatedb::store::WalTableStore; use crate::{CloseReason, Db, ErrorKind, Settings}; use bytes::Bytes; use fail_parallel::fail_point_channel; @@ -179,6 +181,7 @@ mod tests { let object_store: Arc = Arc::new(InMemory::new()); let settings = test_db_options(); let system_clock: Arc = Arc::new(DefaultSystemClock::new()); + let fp_registry = Arc::new(FailPointRegistry::new()); let manifest_store = Arc::new(ManifestStore::new(&Path::from(path), object_store.clone())); let table_store = Arc::new(TableStore::new( @@ -189,6 +192,13 @@ mod tests { TableStoreKind::Main, BlockCachePolicy::default(), )); + let wal_store = Arc::new(WalTableStore::new_with_fp_registry( + Arc::clone(&object_store), + SsTableFormat::default(), + PathResolver::from_root(path), + Arc::clone(&fp_registry), + TableStoreKind::Main, + )); let stored_manifest = StoredManifest::create_new_db( manifest_store.clone(), ManifestCore::new(), @@ -196,7 +206,6 @@ mod tests { ) .await .unwrap(); - let fp_registry = Arc::new(FailPointRegistry::new()); let (fp_tx, event_rx) = fail_point_channel(fp_registry.clone()); let cell = Arc::new(WatchableOnceCell::new()); let recorder = MetricsRecorderHelper::new( @@ -210,7 +219,7 @@ mod tests { let fencer = WriterFencer::new_with_fp_handle( cell.reader(), recorder, - table_store.clone(), + wal_store, &settings, system_clock.clone(), task_executor.clone(), diff --git a/slatedb/src/lib.rs b/slatedb/src/lib.rs index c14fda82dc..3d86763299 100644 --- a/slatedb/src/lib.rs +++ b/slatedb/src/lib.rs @@ -162,6 +162,7 @@ mod single_flight; mod snapshot_manager; mod sorted_run_iterator; mod sst_builder; +mod sst_io; mod sst_iter; mod sst_reader; mod sst_stats; diff --git a/slatedb/src/sst_io.rs b/slatedb/src/sst_io.rs new file mode 100644 index 0000000000..54262b3aa5 --- /dev/null +++ b/slatedb/src/sst_io.rs @@ -0,0 +1,111 @@ +use std::ops::Range; +use std::sync::Arc; + +use bytes::Bytes; +use log::warn; +use object_store::path::Path; +use object_store::{Extensions, GetOptions, GetRange, ObjectStore}; + +use crate::blob::ReadOnlyBlob; +use crate::error::SlateDBError; +use crate::object_store_tag::ObjectStoreCallTag; + +/// Reads one SST object with validation retry while attaching the supplied +/// object-store call tag to every attempt. +macro_rules! read_obj { + ($object_store:expr, $path:expr, $tag:expr, |$obj:ident| $read:expr) => {{ + let object_store = $object_store; + let path = $path; + $crate::sst_io::read_with_validation_retry($tag, move |tag| { + let object_store = object_store.clone(); + let path = path.clone(); + async move { + let $obj = $crate::sst_io::ReadOnlyObject { + object_store, + path, + tag, + }; + $read.await.map_err(|error| error.with_path(&$obj.path)) + } + }) + }}; +} + +pub(crate) use read_obj; + +/// An object-store object exposed through the read-only interface consumed by +/// the shared SST format decoder. +pub(crate) struct ReadOnlyObject { + pub(crate) object_store: Arc, + pub(crate) path: Path, + pub(crate) tag: ObjectStoreCallTag, +} + +impl ReadOnlyObject { + fn extensions(&self) -> Extensions { + self.tag.into() + } +} + +impl ReadOnlyBlob for ReadOnlyObject { + async fn len(&self) -> Result { + let opts = GetOptions { + head: true, + extensions: self.extensions(), + ..GetOptions::default() + }; + let result = self.object_store.get_opts(&self.path, opts).await?; + Ok(result.meta.size) + } + + async fn read_range(&self, range: Range) -> Result { + let opts = GetOptions { + range: Some(GetRange::Bounded(range)), + extensions: self.extensions(), + ..GetOptions::default() + }; + let result = self.object_store.get_opts(&self.path, opts).await?; + Ok(result.bytes().await?) + } + + async fn read(&self) -> Result { + let opts = GetOptions { + extensions: self.extensions(), + ..GetOptions::default() + }; + let result = self.object_store.get_opts(&self.path, opts).await?; + Ok(result.bytes().await?) + } +} + +/// Number of additional attempts after an SST read fails validation. +pub(crate) const MAX_VALIDATION_RETRIES: usize = 1; + +/// Reissues recoverable validation failures with a retry reason on the object +/// store call tag. Caching object-store wrappers use that reason to invalidate +/// a corrupt local copy before the retry. +pub(crate) async fn read_with_validation_retry( + mut tag: ObjectStoreCallTag, + mut read: impl FnMut(ObjectStoreCallTag) -> Fut, +) -> Result +where + Fut: std::future::Future>, +{ + for _ in 0..MAX_VALIDATION_RETRIES { + let result = read(tag).await; + match result { + Err(ref err) => match err.maybe_validation_retry_reason() { + Some(reason) => { + warn!( + "retrying SST read after validation failure [reason={:?}, error={}]", + reason, err + ); + tag.retry = Some(reason); + } + None => return result, + }, + Ok(_) => return result, + } + } + read(tag).await +} diff --git a/slatedb/src/tablestore.rs b/slatedb/src/tablestore.rs index 4d1b6bbf4d..7d8f75b425 100644 --- a/slatedb/src/tablestore.rs +++ b/slatedb/src/tablestore.rs @@ -8,15 +8,12 @@ use futures::{future::join_all, StreamExt}; use log::{debug, warn}; use object_store::buffered::BufWriter; use object_store::path::Path; -use object_store::{ - Extensions, GetOptions, GetRange, ObjectStore, ObjectStoreExt, PutMode, PutOptions, -}; +use object_store::{GetOptions, ObjectStore, ObjectStoreExt, PutMode, PutOptions}; use slatedb_common::object_metadata::IdentifiedObjectMetadata; use slatedb_common::ObjectMetadata; use tokio::io::AsyncWriteExt; use ulid::Ulid; -use crate::blob::ReadOnlyBlob; use crate::block_cache_policy::{should_cache_data_block, BlockCachePolicy}; use crate::db_cache::CacheTarget; use crate::db_cache::{CacheLoader, CachedEntry, CachedKey, DbCache, EncodedCachedFilter}; @@ -32,6 +29,9 @@ pub(crate) use crate::object_store_tag::TableStoreKind; use crate::object_stores::{ObjectStoreType, ObjectStores}; use crate::paths::PathResolver; use crate::sst_builder::EncodedSsTableBuilder; +#[cfg(test)] +use crate::sst_io::MAX_VALIDATION_RETRIES; +use crate::sst_io::{read_obj as read_sst_obj, read_with_validation_retry, ReadOnlyObject}; use crate::sst_stats::SstStats; use crate::types::RowEntry; use crate::wal::slatedb::sst_builder::EncodedWalSsTableBuilder; @@ -50,78 +50,6 @@ pub(crate) struct TableStore { kind: TableStoreKind, } -struct ReadOnlyObject { - object_store: Arc, - path: Path, - tag: ObjectStoreCallTag, -} - -impl ReadOnlyObject { - fn extensions(&self) -> Extensions { - self.tag.into() - } -} - -/// Reads from a [`ReadOnlyObject`] for an SST `$id`, with validation-retry. -/// -/// It expands to the retry-wrapper future, so callers `.await` it. -/// This is used instead of repeating the same retry logic for every individual -/// read from an SST object. -macro_rules! read_obj { - ($store:expr, $id:expr, |$obj:ident| $read:expr) => {{ - let object_store = $store.object_stores.store_for($id); - let path = $store.path($id); - read_with_validation_retry( - ObjectStoreCallTag::new($store.kind, SstType::from($id)), - move |tag| { - let object_store = object_store.clone(); - let path = path.clone(); - async move { - let $obj = ReadOnlyObject { - object_store, - path, - tag, - }; - $read.await.map_err(|e| e.with_path(&$obj.path)) - } - }, - ) - }}; -} - -impl ReadOnlyBlob for ReadOnlyObject { - async fn len(&self) -> Result { - let opts = GetOptions { - head: true, - extensions: self.extensions(), - ..GetOptions::default() - }; - let result = self.object_store.get_opts(&self.path, opts).await?; - Ok(result.meta.size) - } - - async fn read_range(&self, range: Range) -> Result { - let opts = GetOptions { - range: Some(GetRange::Bounded(range)), - extensions: self.extensions(), - ..GetOptions::default() - }; - let result = self.object_store.get_opts(&self.path, opts).await?; - let bytes = result.bytes().await?; - Ok(bytes) - } - - async fn read(&self) -> Result { - let opts = GetOptions { - extensions: self.extensions(), - ..GetOptions::default() - }; - let result = self.object_store.get_opts(&self.path, opts).await?; - let bytes = result.bytes().await?; - Ok(bytes) - } -} - impl TableStore { pub(crate) fn new>( object_stores: ObjectStores, @@ -179,6 +107,7 @@ impl TableStore { /// Relies on the fencing protocol's contiguity invariant: "id exists" is /// monotone-decreasing in id, so binary search is sound. Total HEAD count /// is `O(log N)` for a gap of size N, vs `O(N)` for a windowed scan. + #[allow(unused)] pub(crate) async fn last_seen_wal_id(&self, start_after: u64) -> Result { fail_point!(Arc::clone(&self.fp_registry), "probe-wal-ssts", |_| { Err(SlateDBError::from(std::io::Error::other("oops"))) @@ -315,6 +244,7 @@ impl TableStore { Ok(wal_list) } + #[allow(unused)] pub(crate) async fn next_wal_sst_id( &self, wal_id_last_compacted: u64, @@ -343,6 +273,7 @@ impl TableStore { self.sst_format.table_builder() } + #[allow(unused)] pub(crate) fn wal_table_builder(&self) -> EncodedWalSsTableBuilder { self.sst_format.wal_table_builder() } @@ -475,6 +406,7 @@ impl TableStore { /// /// Uses create-if-absent semantics so any existing WAL object at this ID /// fences the writer by returning [`SlateDBError::Fenced`]. + #[allow(unused)] pub(crate) async fn write_wal_fence(&self, wal_id: u64) -> Result<(), SlateDBError> { let id = SsTableId::Wal(wal_id); fail_point!(self.fp_registry.clone(), "write-wal-sst-io-error", |_| { @@ -608,15 +540,25 @@ impl TableStore { } pub(crate) async fn open_sst(&self, id: &SsTableId) -> Result { - let (info, version) = - read_obj!(self, id, |obj| self.sst_format.read_info_and_version(&obj)).await?; + let (info, version) = read_sst_obj!( + self.object_stores.store_for(id), + self.path(id), + ObjectStoreCallTag::new(self.kind, SstType::from(id)), + |obj| self.sst_format.read_info_and_version(&obj) + ) + .await?; Ok(SsTableHandle::new(*id, version, info)) } #[cfg(test)] pub(crate) async fn read_sst_version(&self, id: &SsTableId) -> Result { - let (_, version) = - read_obj!(self, id, |obj| self.sst_format.read_info_and_version(&obj)).await?; + let (_, version) = read_sst_obj!( + self.object_stores.store_for(id), + self.path(id), + ObjectStoreCallTag::new(self.kind, SstType::from(id)), + |obj| self.sst_format.read_info_and_version(&obj) + ) + .await?; Ok(version) } @@ -670,9 +612,12 @@ impl TableStore { } } } - read_obj!(self, &handle.id, |obj| self - .sst_format - .read_filters(&handle.info, &obj)) + read_sst_obj!( + self.object_stores.store_for(&handle.id), + self.path(&handle.id), + ObjectStoreCallTag::new(self.kind, SstType::from(&handle.id)), + |obj| self.sst_format.read_filters(&handle.info, &obj) + ) .await } @@ -703,9 +648,12 @@ impl TableStore { return Ok(Some(stats.as_ref().clone())); } } - read_obj!(self, &handle.id, |obj| self - .sst_format - .read_stats(&handle.info, &obj)) + read_sst_obj!( + self.object_stores.store_for(&handle.id), + self.path(&handle.id), + ObjectStoreCallTag::new(self.kind, SstType::from(&handle.id)), + |obj| self.sst_format.read_stats(&handle.info, &obj) + ) .await } @@ -734,9 +682,12 @@ impl TableStore { return Ok(index); } } - let index = read_obj!(self, &handle.id, |obj| self - .sst_format - .read_index(&handle.info, &obj)) + let index = read_sst_obj!( + self.object_stores.store_for(&handle.id), + self.path(&handle.id), + ObjectStoreCallTag::new(self.kind, SstType::from(&handle.id)), + |obj| self.sst_format.read_index(&handle.info, &obj) + ) .await?; Ok(Arc::new(index)) } @@ -1059,12 +1010,17 @@ impl TableStore { handle: &SsTableHandle, block: usize, ) -> Result { - read_obj!(self, &handle.id, |obj| async { - let index = self.sst_format.read_index(&handle.info, &obj).await?; - self.sst_format - .read_block(&handle.info, &index, block, &obj) - .await - }) + read_sst_obj!( + self.object_stores.store_for(&handle.id), + self.path(&handle.id), + ObjectStoreCallTag::new(self.kind, SstType::from(&handle.id)), + |obj| async { + let index = self.sst_format.read_index(&handle.info, &obj).await?; + self.sst_format + .read_block(&handle.info, &index, block, &obj) + .await + } + ) .await } @@ -1081,6 +1037,7 @@ impl TableStore { .estimate_encoded_size_compacted(num_entries, size_entries) } + #[allow(unused)] pub(crate) fn estimate_encoded_size_wal( &self, num_entries: usize, @@ -1155,6 +1112,7 @@ impl TableStore { } } +#[allow(unused)] async fn wal_object_exists( object_store: &Arc, path: &Path, @@ -1166,45 +1124,6 @@ async fn wal_object_exists( } } -/// Number of additional attempts after an SST read fails validation. The -/// reissue carries a [`RetryReason`](crate::error::RetryReason) so a caching -/// wrapper drops its local copy. -const MAX_VALIDATION_RETRIES: usize = 1; - -/// Runs `read` with the source/type `tag`, reissuing it with a -/// [`RetryReason`](crate::error::RetryReason) set on the tag when the result is -/// a recoverable validation failure. -/// -/// This is done to enable object store wrappers like a cache to know when -/// to drop a cached entry that failed validation and retry the read from the -/// source of truth (object store) instead of repeatedly returning the same -/// invalid cached entry. -async fn read_with_validation_retry( - mut tag: ObjectStoreCallTag, - mut read: impl FnMut(ObjectStoreCallTag) -> Fut, -) -> Result -where - Fut: std::future::Future>, -{ - for _ in 0..MAX_VALIDATION_RETRIES { - let result = read(tag).await; - match result { - Err(ref err) => match err.maybe_validation_retry_reason() { - Some(reason) => { - warn!( - "retrying SST read after validation failure [reason={:?}, error={}]", - reason, err - ); - tag.retry = Some(reason); - } - None => return result, - }, - Ok(_) => return result, - } - } - read(tag).await -} - /// Builds a [`BufWriter`] whose upload carries `tag` in its extensions. fn tagged_buf_writer( object_store: Arc, diff --git a/slatedb/src/wal/slatedb/iterator.rs b/slatedb/src/wal/slatedb/iterator.rs index aec27d17b3..c643e9cb59 100644 --- a/slatedb/src/wal/slatedb/iterator.rs +++ b/slatedb/src/wal/slatedb/iterator.rs @@ -9,18 +9,18 @@ use tokio::sync::watch; use tokio::task; use tokio::task::JoinHandle; -use crate::db_state::SsTableId; use crate::db_status::DbStatus; use crate::error::SlateDBError; use crate::iter::{EmptyIterator, RowEntryIterator}; use crate::manifest::store::ManifestStore; -use crate::manifest::{SsTableView, VersionedManifest}; -use crate::sst_iter::{SstIterator, SstIteratorOptions}; -use crate::tablestore::TableStore; +use crate::manifest::VersionedManifest; use crate::utils::panic_string; use crate::wal::{WalError, WalIterator as WalIteratorTrait, WalRows}; use crate::RowEntry; +use super::sst_iterator::{WalSstIterator, WalSstIteratorOptions}; +use super::store::WalTableStore; + #[async_trait] pub(crate) trait ManifestReader: Send + Sync + 'static { async fn manifest(&self) -> Result; @@ -41,31 +41,45 @@ impl ManifestReader for ManifestStore { } pub(crate) struct SlateDbWalIteratorOptions { - /// The number of SSTs to preload while replaying + /// The number of WAL SSTs to preload while replaying. pub(crate) sst_batch_size: usize, - /// Options to pass through to underlying SST iterators - pub(crate) sst_iter_options: SstIteratorOptions, + /// Options to pass through to the underlying WAL SST iterators. + pub(crate) sst_iter_options: WalSstIteratorOptions, } impl Default for SlateDbWalIteratorOptions { fn default() -> Self { Self { sst_batch_size: 4, - sst_iter_options: SstIteratorOptions::default(), + sst_iter_options: WalSstIteratorOptions::default(), + } + } +} + +enum WalFileIterator { + Empty(EmptyIterator), + Sst(Box), +} + +impl WalFileIterator { + async fn next(&mut self) -> Result, SlateDBError> { + match self { + Self::Empty(iter) => iter.next().await, + Self::Sst(iter) => iter.next().await, } } } struct WalRowsCollector { wal_id: u64, - iter: Box, + iter: WalFileIterator, rows: Vec, drained: bool, } impl WalRowsCollector { - fn new(wal_id: u64, iter: Box) -> Self { + fn new(wal_id: u64, iter: WalFileIterator) -> Self { Self { wal_id, iter, @@ -144,19 +158,18 @@ impl CurrentWalFile { } /// Iterates over the writes in a range of WAL files, preloading up to -/// `sst_batch_size` WAL SSTs concurrently. Returns the rows of one WAL file per -/// [`WalRows`], and verifies that files carry strictly increasing seq +/// `sst_batch_size` WAL SST handles concurrently. Returns the rows of one WAL +/// file per [`WalRows`], and verifies that files carry strictly increasing seq /// ranges — the ordering callers rely on to split and tag memtables safely. /// -/// Preloading only opens each WAL SST (footer, index, and any eagerly fetched -/// blocks); a file's rows are read out only when it is returned from +/// A file's rows are read sequentially only when it is returned from /// [`Self::next`], so at most one file's rows are materialized at a time. For an /// unbounded end, open tasks poll their assigned future WAL IDs until the files /// appear or the manifest proves that a missing file was truncated. pub(crate) struct SlateDbWalIterator { options: SlateDbWalIteratorOptions, end_bound: WalIteratorEndBound, - table_store: Arc, + wal_store: Arc, next_files: VecDeque>>, next_wal_id: Option, /// The greatest seq returned so far, used to verify that WAL files arrive @@ -204,7 +217,7 @@ impl SlateDbWalIterator { from_wal_id: u64, to_bound: WalIteratorEndBound, options: SlateDbWalIteratorOptions, - table_store: Arc, + wal_store: Arc, ) -> Result { if options.sst_batch_size < 1 { return Err(SlateDBError::InvalidSSTBatchSize(options.sst_batch_size)); @@ -213,7 +226,7 @@ impl SlateDbWalIterator { Ok(Self { options, end_bound: to_bound, - table_store, + wal_store, next_files: VecDeque::new(), next_wal_id: Some(from_wal_id), last_seq: None, @@ -240,49 +253,36 @@ impl SlateDbWalIterator { async fn try_open_file_iter( wal_id: u64, - sst_iter_options: SstIteratorOptions, - table_store: Arc, + sst_iter_options: WalSstIteratorOptions, + wal_store: Arc, ) -> Result { - let sst = match table_store.open_sst(&SsTableId::Wal(wal_id)).await { + let sst = match wal_store.open_sst(wal_id).await { Ok(sst) => sst, Err(SlateDBError::EmptySSTable) => { // Zero-byte WAL files are fence markers; replay them as empty WALs // so the last replayed WAL ID still advances past the marker. return Ok(WalRowsCollector::new( wal_id, - Box::new(EmptyIterator::new()), + WalFileIterator::Empty(EmptyIterator::new()), )); } Err(err) => return Err(err), }; - let iter = SstIterator::new_owned_initialized( - .., - SsTableView::identity(sst), - Arc::clone(&table_store), - sst_iter_options, - ) - .await?; - // An unbounded, unfiltered scan over a WAL SST always yields an - // iterator. `None` means the file cannot be read, and replay must - // fail rather than silently end early and drop the remaining WALs. - let Some(iter) = iter else { - error!( - "could not construct row iterator over WAL SST. [wal_id={}]", - wal_id - ); - return Err(SlateDBError::InvalidDBState); - }; - Ok(WalRowsCollector::new(wal_id, Box::new(iter))) + let iter = WalSstIterator::new(sst, Arc::clone(&wal_store), sst_iter_options).await?; + Ok(WalRowsCollector::new( + wal_id, + WalFileIterator::Sst(Box::new(iter)), + )) } async fn open_file_iter( wal_id: u64, - sst_iter_options: SstIteratorOptions, - table_store: Arc, + sst_iter_options: WalSstIteratorOptions, + wal_store: Arc, end_bound: WalIteratorEndBound, ) -> Result { loop { - match try_open_file_iter(wal_id, sst_iter_options.clone(), Arc::clone(&table_store)) + match try_open_file_iter(wal_id, sst_iter_options.clone(), Arc::clone(&wal_store)) .await { Ok(iter) => return Ok(iter), @@ -312,38 +312,21 @@ impl SlateDbWalIterator { let handle = task::spawn(open_file_iter( next_wal_id, self.options.sst_iter_options.clone(), - Arc::clone(&self.table_store), + Arc::clone(&self.wal_store), self.end_bound.clone(), )); self.next_files.push_back(handle); true } - /// Spawns file loading in the background and populates the iterator with the results of loading - /// the next wal file. - async fn load_next_file(&mut self) -> Result<(), WalError> { - if self.current_file.initialized() { - return Ok(()); - } - // Populate the pre-load queue first to handle the case where it's initially empty - self.spawn_opens(); - // await a mutable ref to the task so that next remains cancel-safe - // see https://docs.rs/tokio/latest/tokio/task/struct.JoinHandle.html#cancel-safety - let Some(join_handle) = self.next_files.front_mut() else { - self.current_file.finish(); - return Ok(()); - }; - let result = join_handle.await; - self.next_files.pop_front(); - // Refill the preload queue before returning so the iterator starts loading the next file - self.spawn_opens(); + fn open_task_result( + end_bound: &WalIteratorEndBound, + result: Result, task::JoinError>, + ) -> Result { match result { - Ok(result) => { - self.current_file.advance(result?); - Ok(()) - } + Ok(result) => result, Err(join_err) => { - let task_name = format!("wal_replay[end_bound={:?}]", self.end_bound); + let task_name = format!("wal_replay[end_bound={end_bound:?}]"); let msg = if let Ok(panic_err) = join_err.try_into_panic() { format!( "wal_replay task panicked unexpectedly. [task_name={}, panic={}]", @@ -360,6 +343,30 @@ impl SlateDbWalIterator { } } + /// Opens WAL handles in the background and promotes the oldest file when a + /// current file is needed. + async fn load_next_file(&mut self) -> Result<(), WalError> { + if self.current_file.initialized() { + return Ok(()); + } + + // Populate the pre-load queue first to handle the case where it's initially empty + self.spawn_opens(); + // await a mutable ref to the task so that next remains cancel-safe + // see https://docs.rs/tokio/latest/tokio/task/struct.JoinHandle.html#cancel-safety + let Some(join_handle) = self.next_files.front_mut() else { + self.current_file.finish(); + return Ok(()); + }; + let result = join_handle.await; + self.next_files.pop_front(); + // Refill the preload queue before returning so the iterator starts loading the next file + self.spawn_opens(); + self.current_file + .advance(Self::open_task_result(&self.end_bound, result)?); + Ok(()) + } + fn terminate( &mut self, result: Result, WalError>, @@ -368,6 +375,7 @@ impl SlateDbWalIterator { for task in self.next_files.drain(..) { task.abort(); } + self.current_file.collector = None; result } } @@ -444,14 +452,12 @@ mod tests { use slatedb_common::clock::DefaultSystemClock; use super::{SlateDbWalIterator, SlateDbWalIteratorOptions, WalIteratorEndBound}; - use crate::block_cache_policy::BlockCachePolicy; - use crate::db_state::SsTableId; use crate::db_status::DbStatusManager; use crate::format::sst::SsTableFormat; use crate::manifest::{Manifest, ManifestCore, VersionedManifest}; - use crate::object_stores::ObjectStores; - use crate::tablestore::{TableStore, TableStoreKind}; + use crate::object_store_tag::TableStoreKind; use crate::types::RowEntry; + use crate::wal::slatedb::store::WalTableStore; use crate::wal::{WalError, WalIterator as _}; fn versioned_manifest(id: u64, next_wal_id: u64) -> VersionedManifest { @@ -534,10 +540,7 @@ mod tests { poll_interval: Duration::from_millis(10), system_clock: Arc::new(DefaultSystemClock::new()), }, - SlateDbWalIteratorOptions { - sst_batch_size: 2, - ..SlateDbWalIteratorOptions::default() - }, + SlateDbWalIteratorOptions::default(), Arc::clone(&table_store), ) .unwrap(); @@ -637,13 +640,13 @@ mod tests { } } for (index, entries) in wal_entries.into_iter().enumerate() { - let mut builder = table_store.wal_table_builder(); + let mut builder = table_store.table_builder(); for entry in entries { builder.add(entry).await.unwrap(); } let encoded_sst = builder.build().await.unwrap(); table_store - .write_sst(&SsTableId::Wal(index as u64 + 1), &encoded_sst) + .write_sst(index as u64 + 1, &encoded_sst) .await .unwrap(); } @@ -715,16 +718,14 @@ mod tests { assert_eq!(last_consumed_wal_file_id, wal_file_count); } - fn test_table_store() -> Arc { + fn test_table_store() -> Arc { let object_store: Arc = Arc::new(InMemory::new()); let path = Path::from("/tmp/test_kv_store"); - Arc::new(TableStore::new( - ObjectStores::new(object_store.clone(), None), + Arc::new(WalTableStore::new( + object_store, SsTableFormat::default(), path, - None, TableStoreKind::Main, - BlockCachePolicy::default(), )) } } diff --git a/slatedb/src/wal/slatedb/mod.rs b/slatedb/src/wal/slatedb/mod.rs index b0e314d394..2c42583a4c 100644 --- a/slatedb/src/wal/slatedb/mod.rs +++ b/slatedb/src/wal/slatedb/mod.rs @@ -5,5 +5,7 @@ pub(crate) mod gc; pub(crate) mod iterator; pub(crate) mod reader; pub(crate) mod sst_builder; +pub(crate) mod sst_iterator; +pub(crate) mod store; pub(crate) mod writer; pub(crate) mod writer_init; diff --git a/slatedb/src/wal/slatedb/reader.rs b/slatedb/src/wal/slatedb/reader.rs index e0503598a8..b17d5c787f 100644 --- a/slatedb/src/wal/slatedb/reader.rs +++ b/slatedb/src/wal/slatedb/reader.rs @@ -7,33 +7,28 @@ use log::error; use object_store::{path::Path, ObjectStore}; use slatedb_common::clock::{DefaultSystemClock, SystemClock}; -use crate::block_cache_policy::BlockCachePolicy; use crate::db_status::DbStatusManager; use crate::error::SlateDBError; use crate::format::sst::SsTableFormat; -use crate::iter::IterationOrder; use crate::manifest::store::ManifestStore; -use crate::object_stores::ObjectStores; -use crate::sst_iter::SstIteratorOptions; -use crate::tablestore::{TableStore, TableStoreKind}; +use crate::object_store_tag::TableStoreKind; use crate::wal::slatedb::iterator::{ ManifestReader, SlateDbWalIterator, SlateDbWalIteratorOptions, WalIteratorEndBound, }; +use crate::wal::slatedb::store::WalTableStore; use crate::wal::{WalError, WalFileRange, WalIterator, WalReader}; #[derive(Clone, Debug)] pub struct SlateDbWalReaderOptions { - /// The number of SSTs to preload while replaying + /// The number of WAL SST handles to preload while replaying. pub sst_batch_size: usize, - /// The number of fetch tasks to spawn per sst. Defaults to 2 so there is always a fetch - /// pending while the current data is being consumed. + /// Retained for compatibility with the existing WAL reader configuration. pub max_fetch_tasks: usize, /// The target number of bytes to fetch in a single request while iterating over WAL SSTs. - /// Each fetch reads the minimum number of blocks such that the resulting read is at least - /// this size or reaches the end of the file. Callers size this to a whole WAL SST - /// (`l0_sst_size_bytes`) so replay reads each file in one request. + /// Each fetch reads enough whole blocks to meet this target or reach the end of the file. + /// The default is 1 MiB. pub read_ahead_bytes: usize, } @@ -51,15 +46,8 @@ impl From for SlateDbWalIteratorOptions { fn from(options: SlateDbWalReaderOptions) -> Self { Self { sst_batch_size: options.sst_batch_size, - sst_iter_options: SstIteratorOptions { - max_fetch_tasks: options.max_fetch_tasks, + sst_iter_options: super::sst_iterator::WalSstIteratorOptions { target_bytes_to_fetch: options.read_ahead_bytes, - cache_blocks: false, - cache_metadata: false, - eager_spawn: true, - order: IterationOrder::Ascending, - prefix: None, - filter_context: None, }, } } @@ -76,7 +64,7 @@ impl From for SlateDbWalIteratorOptions { /// explicitly configured. pub struct SlateDbWalReaderBuilder { path: Option, - table_store: Option>, + wal_store: Option>, object_store: Option>, wal_object_store: Option>, manifest_reader: Option>, @@ -88,7 +76,7 @@ impl Default for SlateDbWalReaderBuilder { fn default() -> Self { Self { path: None, - table_store: None, + wal_store: None, object_store: None, wal_object_store: None, manifest_reader: None, @@ -110,9 +98,9 @@ impl SlateDbWalReaderBuilder { self } - /// Sets an existing table store for internal construction. - pub(crate) fn with_table_store(mut self, table_store: Arc) -> Self { - self.table_store = Some(table_store); + /// Sets an existing WAL table store for internal construction. + pub(crate) fn with_wal_store(mut self, wal_store: Arc) -> Self { + self.wal_store = Some(wal_store); self } @@ -156,7 +144,7 @@ impl SlateDbWalReaderBuilder { /// /// Returns an invalid-configuration error when the database path or /// primary object store has not been configured. Internal callers may - /// instead provide both a table store and manifest reader. + /// instead provide both a WAL store and manifest reader. pub fn build(self) -> Result { let manifest_reader = match self.manifest_reader { Some(manifest_reader) => manifest_reader, @@ -172,8 +160,8 @@ impl SlateDbWalReaderBuilder { Arc::new(ManifestStore::new(&path, object_store)) } }; - let table_store = match self.table_store { - Some(table_store) => table_store, + let wal_store = match self.wal_store { + Some(wal_store) => wal_store, None => { let Some(object_store) = self.object_store.clone() else { return Err(crate::Error::invalid( @@ -183,18 +171,17 @@ impl SlateDbWalReaderBuilder { let Some(path) = self.path.clone() else { return Err(crate::Error::invalid("must specify db path".to_string())); }; - Arc::new(TableStore::new( - ObjectStores::new(object_store, self.wal_object_store.clone()), + let object_store = self.wal_object_store.unwrap_or(object_store); + Arc::new(WalTableStore::new( + object_store, SsTableFormat::default(), path.clone(), - None, TableStoreKind::Reader, - BlockCachePolicy::default(), )) } }; Ok(SlateDbWalReader { - table_store, + wal_store, manifest_reader, system_clock: self.system_clock, options: self.options, @@ -203,7 +190,7 @@ impl SlateDbWalReaderBuilder { } pub struct SlateDbWalReader { - table_store: Arc, + wal_store: Arc, manifest_reader: Arc, system_clock: Arc, options: SlateDbWalReaderOptions, @@ -211,19 +198,19 @@ pub struct SlateDbWalReader { impl SlateDbWalReader { pub(crate) fn new_with_status_manager( - table_store: Arc, + wal_store: Arc, db_status: &DbStatusManager, system_clock: Arc, options: SlateDbWalReaderOptions, ) -> Self { let manifest_reader: Arc = Arc::new(db_status.subscribe()); SlateDbWalReaderBuilder::new() - .with_table_store(table_store) + .with_wal_store(wal_store) .with_manifest_reader(manifest_reader) .with_system_clock(system_clock) .with_options(options) .build() - .expect("table store and manifest reader initialize a WAL reader") + .expect("WAL store and manifest reader initialize a WAL reader") } } @@ -271,16 +258,13 @@ impl WalReader for SlateDbWalReader { from_wal_id, end_bound, self.options.clone().into(), - Arc::clone(&self.table_store), + Arc::clone(&self.wal_store), )?; Ok(Box::new(iterator)) } async fn last_wal_file_id(&self, replay_after_wal_id: u64) -> Result { - let last = self - .table_store - .last_seen_wal_id(replay_after_wal_id) - .await?; + let last = self.wal_store.last_seen_wal_id(replay_after_wal_id).await?; let manifest = self.manifest_reader.manifest().await?; if last < manifest.core().replay_after_wal_id { return Err(WalError::WalTruncated(last)); @@ -339,25 +323,23 @@ mod tests { } #[test] - fn builder_rejects_table_store_without_manifest_reader() { + fn builder_rejects_wal_store_without_manifest_reader() { let object_store: Arc = Arc::new(InMemory::new()); - let table_store = Arc::new(TableStore::new( - ObjectStores::new(object_store, None), + let wal_store = Arc::new(WalTableStore::new( + object_store, SsTableFormat::default(), Path::from("/table-store-without-manifest-reader"), - None, TableStoreKind::Reader, - BlockCachePolicy::default(), )); assert_invalid_build( - SlateDbWalReaderBuilder::new().with_table_store(table_store), + SlateDbWalReaderBuilder::new().with_wal_store(wal_store), "must specify object store", ); } #[test] - fn builder_rejects_manifest_reader_without_table_store() { + fn builder_rejects_manifest_reader_without_wal_store() { let object_store: Arc = Arc::new(InMemory::new()); let path = Path::from("/manifest-reader-without-table-store"); let manifest_reader: Arc = @@ -370,22 +352,20 @@ mod tests { } #[test] - fn builder_accepts_table_store_and_manifest_reader() { + fn builder_accepts_wal_store_and_manifest_reader() { let object_store: Arc = Arc::new(InMemory::new()); let path = Path::from("/table-store-and-manifest-reader"); - let table_store = Arc::new(TableStore::new( - ObjectStores::new(Arc::clone(&object_store), None), + let wal_store = Arc::new(WalTableStore::new( + Arc::clone(&object_store), SsTableFormat::default(), path.clone(), - None, TableStoreKind::Reader, - BlockCachePolicy::default(), )); let manifest_reader: Arc = Arc::new(ManifestStore::new(&path, object_store)); assert!(SlateDbWalReaderBuilder::new() - .with_table_store(table_store) + .with_wal_store(wal_store) .with_manifest_reader(manifest_reader) .build() .is_ok()); @@ -604,8 +584,8 @@ mod tests { .with_path(Path::from("/reader_normalizes_range_bounds")) .build() .unwrap(); - wal_reader.table_store.write_wal_fence(1).await.unwrap(); - wal_reader.table_store.write_wal_fence(2).await.unwrap(); + wal_reader.wal_store.write_wal_fence(1).await.unwrap(); + wal_reader.wal_store.write_wal_fence(2).await.unwrap(); let mut iterator = wal_reader .iterator(WalFileRange(Bound::Excluded(1), Bound::Included(2))) @@ -693,19 +673,14 @@ mod tests { async fn last_wal_file_id_errors_when_last_id_precedes_manifest_gc_cutoff() { let object_store: Arc = Arc::new(InMemory::new()); let path = Path::from("/last_wal_file_id_before_gc_cutoff"); - let table_store = Arc::new(TableStore::new( - ObjectStores::new(Arc::clone(&object_store), None), + let wal_store = Arc::new(WalTableStore::new( + Arc::clone(&object_store), SsTableFormat::default(), path.clone(), - None, TableStoreKind::Reader, - BlockCachePolicy::default(), )); - table_store - .table_writer(SsTableId::Wal(1)) - .close() - .await - .unwrap(); + let encoded_sst = wal_store.table_builder().build().await.unwrap(); + wal_store.write_sst(1, &encoded_sst).await.unwrap(); let mut core = ManifestCore::new(); core.next_wal_sst_id = 3; diff --git a/slatedb/src/wal/slatedb/sst_iterator.rs b/slatedb/src/wal/slatedb/sst_iterator.rs new file mode 100644 index 0000000000..5e41c56481 --- /dev/null +++ b/slatedb/src/wal/slatedb/sst_iterator.rs @@ -0,0 +1,171 @@ +#![allow(dead_code)] // Implemented ahead of migrating WAL replay to WalTableStore. + +use std::collections::VecDeque; +use std::sync::Arc; + +use super::store::{WalFileHandle, WalTableStore}; +use crate::block_iterator::DataBlockIterator; +use crate::config::SstBlockSize; +use crate::error::SlateDBError; +use crate::flatbuffer_types::SsTableIndexOwned; +use crate::format::block::Block; +use crate::iter::IterationOrder; +use crate::types::RowEntry; + +#[derive(Clone, Debug)] +pub(crate) struct WalSstIteratorOptions { + /// Target encoded bytes per block-fetch request. + pub(crate) target_bytes_to_fetch: usize, +} + +impl Default for WalSstIteratorOptions { + fn default() -> Self { + Self { + target_bytes_to_fetch: SstBlockSize::default().as_bytes(), + } + } +} + +/// Iterates all rows in one WAL SST in ascending sequence order. +/// +/// WAL replay only performs whole-file scans, so this iterator deliberately has +/// no range/view abstraction, filters, cache controls, descending mode, seek, +/// or speculative fetch scheduling. +pub(crate) struct WalSstIterator { + table: WalFileHandle, + index: Arc, + block_iter: Option>>, + next_block_idx_to_fetch: usize, + fetched_blocks: VecDeque>, + table_store: Arc, + options: WalSstIteratorOptions, +} + +impl WalSstIterator { + pub(crate) async fn new( + table: WalFileHandle, + table_store: Arc, + options: WalSstIteratorOptions, + ) -> Result { + assert!(options.target_bytes_to_fetch > 0); + let index = table_store.read_index(&table).await?; + Ok(Self { + table, + index, + block_iter: None, + next_block_idx_to_fetch: 0, + fetched_blocks: VecDeque::new(), + table_store, + options, + }) + } + + /// Returns the next WAL row in ascending sequence order. + pub(crate) async fn next(&mut self) -> Result, SlateDBError> { + loop { + if let Some(iter) = &mut self.block_iter { + if let Some(row) = iter.next().await? { + return Ok(Some(row)); + } + } + if !self.load_next_block().await? { + return Ok(None); + } + } + } + + async fn load_next_block(&mut self) -> Result { + loop { + if let Some(block) = self.fetched_blocks.pop_front() { + self.block_iter = Some(DataBlockIterator::new( + block, + self.table.format_version, + IterationOrder::Ascending, + )?); + return Ok(true); + } + + let num_blocks = self.index.borrow().block_meta().len(); + if self.next_block_idx_to_fetch == num_blocks { + self.block_iter = None; + return Ok(false); + } + + let blocks = self.table_store.block_range_for_target_bytes( + &self.table, + &self.index, + self.next_block_idx_to_fetch, + self.options.target_bytes_to_fetch, + ); + let next_block_idx_to_fetch = blocks.end; + let fetched_blocks = self + .table_store + .read_blocks_using_index(&self.table, Arc::clone(&self.index), blocks) + .await?; + // Commit the cursor only after the read succeeds so cancelling the + // read future cannot cause the next call to skip these blocks. + self.next_block_idx_to_fetch = next_block_idx_to_fetch; + self.fetched_blocks = fetched_blocks; + } + } +} + +#[cfg(test)] +mod tests { + use object_store::memory::InMemory; + use object_store::path::Path; + use object_store::ObjectStore; + + use super::*; + use crate::flatbuffer_types::FlatBufferSsTableInfoCodec; + use crate::format::sst::SsTableFormat; + use crate::object_store_tag::TableStoreKind; + use crate::wal::slatedb::sst_builder::EncodedWalSsTableBuilder; + + fn test_store() -> Arc { + let object_store: Arc = Arc::new(InMemory::new()); + Arc::new(WalTableStore::new( + object_store, + SsTableFormat::default(), + Path::from("test-db"), + TableStoreKind::Main, + )) + } + + #[tokio::test] + async fn should_iterate_the_whole_wal_sequentially() { + let store = test_store(); + let rows: Vec<_> = (1..=6) + .map(|seq| { + RowEntry::new_value( + format!("key-{seq}").as_bytes(), + format!("value-{seq}").as_bytes(), + seq, + ) + }) + .collect(); + let mut builder = + EncodedWalSsTableBuilder::new(32, Box::new(FlatBufferSsTableInfoCodec {})); + for row in rows.iter().cloned() { + builder.add(row).await.unwrap(); + } + let encoded = builder.build().await.unwrap(); + let table = store.write_sst(1, &encoded).await.unwrap(); + let mut iter = WalSstIterator::new( + table, + store, + WalSstIteratorOptions { + target_bytes_to_fetch: 1, + }, + ) + .await + .unwrap(); + + let mut actual = Vec::new(); + while let Some(row) = iter.next().await.unwrap() { + actual.push(row); + } + assert_eq!(actual, rows); + assert!(iter.next().await.unwrap().is_none()); + } +} diff --git a/slatedb/src/wal/slatedb/store.rs b/slatedb/src/wal/slatedb/store.rs new file mode 100644 index 0000000000..8837405f6e --- /dev/null +++ b/slatedb/src/wal/slatedb/store.rs @@ -0,0 +1,447 @@ +#![allow(dead_code)] // This store is intentionally implemented before its call sites are migrated. + +use std::collections::VecDeque; +use std::ops::Range; +use std::sync::Arc; + +use bytes::Bytes; +use fail_parallel::{fail_point, FailPointRegistry}; +use futures::future::join_all; +use log::debug; +use object_store::path::Path; +use object_store::{ObjectStore, ObjectStoreExt, PutMode, PutOptions}; +use serde::Serialize; + +use crate::db_state::{SsTableId, SsTableInfo, SstType}; +use crate::error::SlateDBError; +use crate::flatbuffer_types::SsTableIndexOwned; +use crate::format::block::Block; +use crate::format::sst::{EncodedSsTable, SsTableFormat}; +use crate::object_store_tag::{ObjectStoreCallTag, TableStoreKind}; +use crate::paths::PathResolver; +use crate::sst_io::{read_obj, read_with_validation_retry, ReadOnlyObject}; +use crate::wal::slatedb::sst_builder::EncodedWalSsTableBuilder; + +#[derive(Clone, Copy, Debug, Eq, Hash, Ord, PartialEq, PartialOrd, Serialize)] +pub(crate) struct WalFileId(u64); + +impl WalFileId { + pub(crate) fn value(self) -> u64 { + self.0 + } +} + +impl From for WalFileId { + fn from(value: u64) -> Self { + Self(value) + } +} + +#[derive(Clone, Debug, PartialEq, Serialize)] +pub(crate) struct WalFileHandle { + pub(crate) id: WalFileId, + pub(crate) format_version: u16, + pub(crate) info: SsTableInfo, +} + +impl WalFileHandle { + fn new(id: u64, format_version: u16, info: SsTableInfo) -> Self { + Self { + id: id.into(), + format_version, + info, + } + } +} + +/// Cacheless storage adapter for SlateDB's object-store-backed WAL files. +/// +/// WALs continue to use the shared SST format and builders. This type owns only +/// WAL object-store operations and deliberately has no `DbCache` integration. +pub(crate) struct WalTableStore { + object_store: Arc, + sst_format: SsTableFormat, + path_resolver: PathResolver, + fp_registry: Arc, + kind: TableStoreKind, +} + +impl WalTableStore { + pub(crate) fn new>( + object_store: Arc, + sst_format: SsTableFormat, + root_path: P, + kind: TableStoreKind, + ) -> Self { + Self::new_with_fp_registry( + object_store, + sst_format, + PathResolver::from_root(root_path), + Arc::new(FailPointRegistry::new()), + kind, + ) + } + + pub(crate) fn new_with_fp_registry( + object_store: Arc, + sst_format: SsTableFormat, + path_resolver: PathResolver, + fp_registry: Arc, + kind: TableStoreKind, + ) -> Self { + Self { + object_store, + sst_format, + path_resolver, + fp_registry, + kind, + } + } + + pub(crate) fn table_builder(&self) -> EncodedWalSsTableBuilder { + self.sst_format.wal_table_builder() + } + + pub(crate) fn estimate_encoded_size(&self, num_entries: usize, size_entries: usize) -> usize { + self.sst_format + .estimate_encoded_size_wal(num_entries, size_entries) + } + + /// Writes a WAL SST with create-if-absent semantics required for fencing. + pub(crate) async fn write_sst( + &self, + wal_id: u64, + encoded_sst: &EncodedSsTable, + ) -> Result { + fail_point!(self.fp_registry.clone(), "write-wal-sst-io-error", |_| { + Err(slatedb_io_error()) + }); + + self.write_create(wal_id, encoded_sst.remaining_as_bytes()) + .await?; + Ok(WalFileHandle::new( + wal_id, + encoded_sst.format_version, + encoded_sst.info.clone(), + )) + } + + /// Writes a zero-byte WAL object as a fencing marker. + pub(crate) async fn write_wal_fence(&self, wal_id: u64) -> Result<(), SlateDBError> { + fail_point!(self.fp_registry.clone(), "write-wal-sst-io-error", |_| { + Err(slatedb_io_error()) + }); + self.write_create(wal_id, Bytes::new()).await + } + + async fn write_create(&self, wal_id: u64, data: Bytes) -> Result<(), SlateDBError> { + let path = self.path(wal_id); + let opts = PutOptions { + mode: PutMode::Create, + extensions: ObjectStoreCallTag::new(self.kind, SstType::Wal).into(), + ..PutOptions::default() + }; + self.object_store + .put_opts(&path, data.into(), opts) + .await + .map_err(|error| match error { + object_store::Error::AlreadyExists { .. } => { + debug!("path already exists [path={}]", path); + SlateDBError::Fenced + } + error => SlateDBError::from(error), + })?; + Ok(()) + } + + pub(crate) async fn open_sst(&self, wal_id: u64) -> Result { + let (info, version) = read_obj!( + Arc::clone(&self.object_store), + self.path(wal_id), + ObjectStoreCallTag::new(self.kind, SstType::Wal), + |obj| self.sst_format.read_info_and_version(&obj) + ) + .await?; + Ok(WalFileHandle::new(wal_id, version, info)) + } + + pub(crate) async fn read_index( + &self, + handle: &WalFileHandle, + ) -> Result, SlateDBError> { + let index = read_obj!( + Arc::clone(&self.object_store), + self.path(handle.id.value()), + ObjectStoreCallTag::new(self.kind, SstType::Wal), + |obj| self.sst_format.read_index(&handle.info, &obj) + ) + .await?; + Ok(Arc::new(index)) + } + + pub(crate) fn block_range_for_target_bytes( + &self, + handle: &WalFileHandle, + index: &SsTableIndexOwned, + first_block: usize, + target_bytes: usize, + ) -> Range { + assert!(target_bytes > 0); + + let index = index.borrow(); + let block_meta = index.block_meta(); + let num_blocks = block_meta.len(); + assert!(first_block < num_blocks); + let target_bytes = u64::try_from(target_bytes).unwrap_or(u64::MAX); + + let mut blocks = first_block..first_block + 1; + loop { + let byte_range = self + .sst_format + .block_range(blocks.clone(), &handle.info, &index); + if byte_range.end.saturating_sub(byte_range.start) >= target_bytes + || blocks.end == num_blocks + { + return blocks; + } + blocks.end += 1; + } + } + + pub(crate) fn block_range_size( + &self, + handle: &WalFileHandle, + index: &SsTableIndexOwned, + blocks: Range, + ) -> usize { + if blocks.is_empty() { + return 0; + } + let byte_range = self + .sst_format + .block_range(blocks, &handle.info, &index.borrow()); + usize::try_from(byte_range.end.saturating_sub(byte_range.start)).unwrap_or(usize::MAX) + } + + pub(crate) async fn read_blocks_using_index( + &self, + handle: &WalFileHandle, + index: Arc, + blocks: Range, + ) -> Result>, SlateDBError> { + let object_store = Arc::clone(&self.object_store); + let path = self.path(handle.id.value()); + let index = &index; + let blocks = + read_with_validation_retry(ObjectStoreCallTag::new(self.kind, SstType::Wal), |tag| { + let obj = ReadOnlyObject { + object_store: Arc::clone(&object_store), + path: path.clone(), + tag, + }; + let blocks = blocks.clone(); + async move { + self.sst_format + .read_blocks(&handle.info, index, blocks, &obj) + .await + .map_err(|error| error.with_path(&obj.path)) + } + }) + .await?; + Ok(blocks.into_iter().map(Arc::new).collect()) + } + + /// Find the highest WAL SST id present in the object store at or above + /// `start_after + 1`, returning `start_after` if none exist. + /// + /// `start_after` should be a known lower bound (e.g. `replay_after_wal_id` + /// from the manifest, or the highest already-replayed WAL id). Passing 0 + /// scans the entire WAL id space. + /// + /// Two phases: + /// 1. Parallel exponential probe at offsets `2^0, 2^1, ..., 2^k` from + /// `start_after`. One RTT per round of 8 exponents. Brackets the + /// frontier between two adjacent powers of two. + /// 2. Sequential binary search inside the bracketed range to find the + /// exact frontier. + /// + /// Relies on the fencing protocol's contiguity invariant: "id exists" is + /// monotone-decreasing in id, so binary search is sound. Total HEAD count + /// is `O(log N)` for a gap of size N, vs `O(N)` for a windowed scan. + pub(crate) async fn last_seen_wal_id(&self, start_after: u64) -> Result { + fail_point!(Arc::clone(&self.fp_registry), "probe-wal-ssts", |_| { + Err(SlateDBError::from(std::io::Error::other("oops"))) + }); + + const ROUND_SIZE: u32 = 8; + const MAX_EXP: u32 = 48; + + let mut lo_offset = None; + let mut hi_offset = None; + let mut next_exp = 0; + + while hi_offset.is_none() { + if next_exp >= MAX_EXP { + return Err(SlateDBError::InvalidDBState); + } + let end_exp = (next_exp + ROUND_SIZE).min(MAX_EXP); + let exps: Vec = (next_exp..end_exp).collect(); + let probes = exps.iter().map(|&exp| { + let offset = 1u64 << exp; + let path = self.path(start_after + offset); + let object_store = Arc::clone(&self.object_store); + async move { wal_object_exists(&object_store, &path).await } + }); + let results = join_all(probes).await; + + for (exp, result) in exps.iter().zip(results) { + let offset = 1u64 << exp; + if result? { + lo_offset = Some(offset); + } else { + hi_offset = Some(offset); + break; + } + } + next_exp = end_exp; + } + + let hi = hi_offset.expect("loop exits only after finding an upper bound"); + let Some(lo) = lo_offset else { + return Ok(start_after); + }; + + let mut left = lo + 1; + let mut right = hi; + while left < right { + let mid = left + (right - left) / 2; + if wal_object_exists(&self.object_store, &self.path(start_after + mid)).await? { + left = mid + 1; + } else { + right = mid; + } + } + Ok(start_after + left - 1) + } + + pub(crate) async fn next_wal_sst_id( + &self, + wal_id_last_compacted: u64, + ) -> Result { + Ok(self.last_seen_wal_id(wal_id_last_compacted).await? + 1) + } + + fn path(&self, wal_id: u64) -> Path { + self.path_resolver.sst_path(&SsTableId::Wal(wal_id)) + } +} + +async fn wal_object_exists( + object_store: &Arc, + path: &Path, +) -> Result { + match object_store.head(path).await { + Ok(_) => Ok(true), + Err(object_store::Error::NotFound { .. }) => Ok(false), + Err(error) => Err(SlateDBError::from(error)), + } +} + +fn slatedb_io_error() -> SlateDBError { + SlateDBError::from(std::io::Error::other("oops")) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::block_iterator::DataBlockIterator; + use crate::iter::IterationOrder; + use crate::types::RowEntry; + use object_store::memory::InMemory; + + fn test_store() -> WalTableStore { + let object_store: Arc = Arc::new(InMemory::new()); + WalTableStore::new( + object_store, + SsTableFormat::default(), + Path::from("test-db"), + TableStoreKind::Main, + ) + } + + #[tokio::test] + async fn writes_and_reads_wal_sst_without_a_cache() { + let store = test_store(); + let rows = [ + RowEntry::new_value(b"first", b"value-1", 10), + RowEntry::new_value(b"second", b"value-2", 11), + ]; + assert!(store.estimate_encoded_size(2, 26) > 0); + + let mut builder = store.table_builder(); + for row in rows.iter().cloned() { + builder.add(row).await.unwrap(); + } + let encoded = builder.build().await.unwrap(); + let written = store.write_sst(1, &encoded).await.unwrap(); + let opened = store.open_sst(1).await.unwrap(); + + assert_eq!(written, opened); + assert_eq!(opened.id.value(), 1); + + let index = store.read_index(&opened).await.unwrap(); + let block_count = index.borrow().block_meta().len(); + assert_eq!(block_count, 1); + assert_eq!( + index.borrow().block_meta().get(0).first_key().bytes(), + &10u64.to_be_bytes() + ); + + let block_range = store.block_range_for_target_bytes(&opened, &index, 0, usize::MAX); + assert_eq!(block_range, 0..block_count); + assert!(store.block_range_size(&opened, &index, block_range.clone()) > 0); + + let blocks = store + .read_blocks_using_index(&opened, Arc::clone(&index), block_range) + .await + .unwrap(); + let mut actual = Vec::new(); + for block in blocks { + let mut iter = + DataBlockIterator::new(block, opened.format_version, IterationOrder::Ascending) + .unwrap(); + while let Some(row) = iter.next().await.unwrap() { + actual.push(row); + } + } + assert_eq!(actual, rows); + } + + #[tokio::test] + async fn uses_create_semantics_for_wals_and_fences() { + let store = test_store(); + + store.write_wal_fence(1).await.unwrap(); + assert_eq!(store.last_seen_wal_id(0).await.unwrap(), 1); + assert_eq!(store.next_wal_sst_id(0).await.unwrap(), 2); + assert!(matches!( + store.write_wal_fence(1).await, + Err(SlateDBError::Fenced) + )); + assert!(matches!( + store.open_sst(1).await, + Err(SlateDBError::EmptySSTable) + )); + + let mut builder = store.table_builder(); + builder + .add(RowEntry::new_value(b"key", b"value", 12)) + .await + .unwrap(); + let encoded = builder.build().await.unwrap(); + assert!(matches!( + store.write_sst(1, &encoded).await, + Err(SlateDBError::Fenced) + )); + } +} diff --git a/slatedb/src/wal/slatedb/writer.rs b/slatedb/src/wal/slatedb/writer.rs index 4cebb1216c..cd0beed991 100644 --- a/slatedb/src/wal/slatedb/writer.rs +++ b/slatedb/src/wal/slatedb/writer.rs @@ -5,10 +5,8 @@ use std::sync::Arc; use std::time::Duration; use self::stats::WalBufferStats; -use crate::db_state::SsTableId; use crate::dispatcher::{MessageHandler, MessageHandlerExecutor, MessageTickerDef}; use crate::error::SlateDBError; -use crate::tablestore::TableStore; use crate::types::RowEntry; use crate::utils::SafeSender; use crate::utils::{format_bytes_si, WatchableOnceCellReader}; @@ -21,6 +19,8 @@ use slatedb_common::metrics::MetricsRecorderHelper; use tokio::{runtime::Handle, sync::oneshot}; use tracing::instrument; +use super::store::WalTableStore; + pub(crate) const WAL_BUFFER_TASK_NAME: &str = "wal_writer"; /// [`SlateDbWalWriter`] buffers write operations in memory before flushing them to persistent storage. @@ -51,7 +51,7 @@ pub(crate) const WAL_BUFFER_TASK_NAME: &str = "wal_writer"; pub(crate) struct SlateDbWalWriter { inner: Arc>, stats: Arc, - table_store: Arc, + table_store: Arc, max_wal_bytes_size: usize, max_wal_flushes_before_l0_flush: u64, /// The largest flush_epoch for which a size-triggered flush request has been @@ -112,7 +112,7 @@ impl SlateDbWalWriter { closed_result_reader: WatchableOnceCellReader>, recorder: &MetricsRecorderHelper, last_flushed_wal_id: u64, - table_store: Arc, + table_store: Arc, max_wal_bytes_size: usize, max_wal_flushes_before_l0_flush: u64, max_flush_interval: Option, @@ -282,10 +282,10 @@ impl SlateDbWalWriterInner { Ok(()) } - fn needs_flush(&self, table_store: &TableStore, max_wal_bytes_size: usize) -> (bool, u64) { + fn needs_flush(&self, table_store: &WalTableStore, max_wal_bytes_size: usize) -> (bool, u64) { // check the size of the current wal let current_wal_size = - table_store.estimate_encoded_size_wal(self.current_wal.len(), self.current_wal.size()); + table_store.estimate_encoded_size(self.current_wal.len(), self.current_wal.size()); trace!( "checking flush trigger [current_wal_size={}, max_wal_bytes_size={}]", format_bytes_si(current_wal_size as u64), @@ -306,18 +306,18 @@ impl SlateDbWalWriterInner { } /// Returns the total size of all unflushed WALs in bytes. - fn estimated_bytes(&self, table_store: &TableStore) -> usize { + fn estimated_bytes(&self, table_store: &WalTableStore) -> usize { let current_wal_size = - table_store.estimate_encoded_size_wal(self.current_wal.len(), self.current_wal.size()); + table_store.estimate_encoded_size(self.current_wal.len(), self.current_wal.size()); let imm_wal_size = self .immutable_wals .iter() - .map(|(_, wal)| table_store.estimate_encoded_size_wal(wal.len(), wal.size())) + .map(|(_, wal)| table_store.estimate_encoded_size(wal.len(), wal.size())) .sum::(); current_wal_size + imm_wal_size } - fn status(&self, table_store: &TableStore) -> Result { + fn status(&self, table_store: &WalTableStore) -> Result { let status = self.compute_status(table_store); if status.closed_reason.is_none() { Ok(status) @@ -326,7 +326,7 @@ impl SlateDbWalWriterInner { } } - fn compute_status(&self, table_store: &TableStore) -> WalStatus { + fn compute_status(&self, table_store: &WalTableStore) -> WalStatus { let flushing_wal_entries_count = self .immutable_wals .iter() @@ -342,7 +342,7 @@ impl SlateDbWalWriterInner { } } - fn mark_closed(&mut self, reason: WalError, table_store: &TableStore) -> WalStatus { + fn mark_closed(&mut self, reason: WalError, table_store: &WalTableStore) -> WalStatus { self.flush_task_exited_reason = Some(reason); self.freeze_current_wal(); self.immutable_wals.clear(); @@ -467,7 +467,7 @@ impl Debug for WalFlushWork { struct WalFlushHandler { max_flush_interval: Option, inner: Arc>, - table_store: Arc, + table_store: Arc, stats: Arc, listener: Option, } @@ -511,7 +511,7 @@ impl WalFlushHandler { async fn do_flush_one_wal(&self, wal_id: u64, wal: Arc) -> Result<(), SlateDBError> { self.stats.flushes.increment(1); - let mut sst_builder = self.table_store.wal_table_builder(); + let mut sst_builder = self.table_store.table_builder(); let mut iter = wal.iter(); while let Some(entry) = iter.next() { sst_builder.add(entry).await?; @@ -519,9 +519,7 @@ impl WalFlushHandler { let encoded_sst = sst_builder.build().await?; let written_bytes = encoded_sst.remaining_len() as u64; - self.table_store - .write_sst(&SsTableId::Wal(wal_id), &encoded_sst) - .await?; + self.table_store.write_sst(wal_id, &encoded_sst).await?; self.stats.flush_bytes.increment(written_bytes); Ok(()) } @@ -603,7 +601,7 @@ impl MessageHandler for WalFlushHandler { #[derive(Clone)] struct SlateDbWalObserver { inner: Arc>, - table_store: Arc, + table_store: Arc, } impl wal::WalObserver for SlateDbWalObserver { @@ -656,16 +654,12 @@ pub mod stats { #[cfg(test)] mod tests { use super::*; - use crate::block_cache_policy::BlockCachePolicy; use crate::db_status::{ClosedResultWriter, DbStatusManager}; use crate::format::sst::SsTableFormat; - use crate::iter::RowEntryIterator; - use crate::manifest::SsTableView; - use crate::object_stores::ObjectStores; + use crate::object_store_tag::TableStoreKind; use crate::oracle::DbOracle; - use crate::sst_iter::{SstIterator, SstIteratorOptions}; - use crate::tablestore::{TableStore, TableStoreKind}; use crate::types::{RowEntry, ValueDeletable}; + use crate::wal::slatedb::sst_iterator::{WalSstIterator, WalSstIteratorOptions}; use bytes::Bytes; use object_store::{memory::InMemory, path::Path, ObjectStore}; use slatedb_common::clock::DefaultSystemClock; @@ -802,7 +796,7 @@ mod tests { async fn setup_wal_buffer() -> ( SlateDbWalWriter, - Arc, + Arc, Arc, Arc, ) { @@ -813,7 +807,7 @@ mod tests { flush_interval: Duration, ) -> ( SlateDbWalWriter, - Arc, + Arc, Arc, Arc, ) { @@ -825,18 +819,16 @@ mod tests { listener: wal::WalStatusListener, ) -> ( SlateDbWalWriter, - Arc, + Arc, Arc, Arc, ) { let object_store: Arc = Arc::new(InMemory::new()); - let table_store = Arc::new(TableStore::new( - ObjectStores::new(object_store, None), + let table_store = Arc::new(WalTableStore::new( + object_store, SsTableFormat::default(), Path::from("/root"), - None, TableStoreKind::Main, - BlockCachePolicy::default(), )); let system_clock = Arc::new(DefaultSystemClock::new()); let status_manager = Arc::new(DbStatusManager::new(0)); @@ -909,19 +901,11 @@ mod tests { wal_buffer.flush().await.unwrap().await.unwrap(); // Verify entries were written to storage - let sst_iter_options = SstIteratorOptions { - eager_spawn: true, - ..SstIteratorOptions::default() - }; - let mut iter = SstIterator::new_owned_initialized( - .., - SsTableView::identity(table_store.open_sst(&SsTableId::Wal(1)).await.unwrap()), - table_store.clone(), - sst_iter_options, - ) - .await - .unwrap() - .unwrap(); + let table = table_store.open_sst(1).await.unwrap(); + let mut iter = + WalSstIterator::new(table, table_store.clone(), WalSstIteratorOptions::default()) + .await + .unwrap(); let read_entry1 = iter.next().await.unwrap().unwrap(); assert_eq!(read_entry1.key, entry1.key); diff --git a/slatedb/src/wal/slatedb/writer_init.rs b/slatedb/src/wal/slatedb/writer_init.rs index 84a3dc7042..44e1cfc054 100644 --- a/slatedb/src/wal/slatedb/writer_init.rs +++ b/slatedb/src/wal/slatedb/writer_init.rs @@ -1,7 +1,6 @@ use crate::dispatcher::MessageHandlerExecutor; use crate::error::SlateDBError; use crate::manifest::Manifest; -use crate::tablestore::TableStore; use crate::utils::WatchableOnceCellReader; use crate::wal::slatedb::iterator::{SlateDbWalIterator, WalIteratorEndBound}; use crate::wal::slatedb::reader::SlateDbWalReaderOptions; @@ -14,6 +13,8 @@ use slatedb_common::metrics::MetricsRecorderHelper; use std::sync::Arc; use std::time::Duration; +use super::store::WalTableStore; + #[derive(Clone, Copy)] pub(crate) struct SlateDbWalWriterInitOptions { max_wal_bytes_size: usize, @@ -34,7 +35,7 @@ impl From<&Settings> for SlateDbWalWriterInitOptions { pub(crate) struct SlateDbWalWriterInit { closed_result_reader: WatchableOnceCellReader>, recorder: MetricsRecorderHelper, - table_store: Arc, + table_store: Arc, max_wal_bytes_size: usize, max_wal_flushes_before_l0_flush: u64, max_flush_interval: Option, @@ -48,7 +49,7 @@ impl SlateDbWalWriterInit { pub(crate) async fn load( closed_result_reader: WatchableOnceCellReader>, recorder: MetricsRecorderHelper, - table_store: Arc, + table_store: Arc, options: SlateDbWalWriterInitOptions, manifest: &Manifest, task_executor: Arc, diff --git a/slatedb/src/wal_replay.rs b/slatedb/src/wal_replay.rs index 0344e834fd..d936b17e11 100644 --- a/slatedb/src/wal_replay.rs +++ b/slatedb/src/wal_replay.rs @@ -6,6 +6,8 @@ use crate::tablestore::TableStore; use crate::wal::slatedb::iterator::{ SlateDbWalIterator, SlateDbWalIteratorOptions, WalIteratorEndBound, }; +#[cfg(test)] +use crate::wal::slatedb::store::WalTableStore; use crate::wal::WalIterator as WalIteratorTrait; #[cfg(test)] use std::ops::Range; @@ -61,12 +63,13 @@ impl WalReplayIterator { iterator_options: SlateDbWalIteratorOptions, replay_options: WalReplayOptions, table_store: Arc, + wal_store: Arc, ) -> Result { let wal_iter = SlateDbWalIterator::range( wal_id_range.start, WalIteratorEndBound::Exclusive(wal_id_range.end), iterator_options, - Arc::clone(&table_store), + wal_store, )?; Self::for_wal_iterator(Box::new(wal_iter), db_state, replay_options, table_store) } @@ -189,7 +192,6 @@ mod tests { use super::{SlateDbWalIteratorOptions, WalReplayIterator, WalReplayOptions}; use crate::block_cache_policy::BlockCachePolicy; use crate::bytes_range::BytesRange; - use crate::db_state::SsTableId; use crate::format::sst::SsTableFormat; use crate::iter::{IterationOrder, RowEntryIterator}; use crate::manifest::ManifestCore; @@ -198,6 +200,7 @@ mod tests { use crate::proptest_util::{rng, sample}; use crate::tablestore::{TableStore, TableStoreKind}; use crate::types::RowEntry; + use crate::wal::slatedb::store::WalTableStore; use crate::wal::{WalError, WalIterator as WalIteratorTrait, WalRows}; use crate::{error::SlateDBError, test_utils}; use async_trait::async_trait; @@ -228,9 +231,10 @@ mod tests { db_state: &ManifestCore, options: WalReplayOptions, table_store: Arc, + wal_store: Arc, ) -> Result { let wal_id_start = db_state.replay_after_wal_id + 1; - let wal_id_end = table_store + let wal_id_end = wal_store .last_seen_wal_id(db_state.replay_after_wal_id) .await?; let wal_id_range = wal_id_start..(wal_id_end + 1); @@ -240,6 +244,7 @@ mod tests { SlateDbWalIteratorOptions::default(), options, table_store, + wal_store, ) } } @@ -360,12 +365,13 @@ mod tests { #[tokio::test] async fn should_replay_empty_wal() { - let table_store = test_table_store(); - write_empty_wal(1, Arc::clone(&table_store)).await.unwrap(); + let (table_store, wal_store) = test_stores(); + write_empty_wal(1, Arc::clone(&wal_store)).await.unwrap(); let mut replay_iter = WalReplayIterator::all_wal_ids( &ManifestCore::new(), WalReplayOptions::default(), Arc::clone(&table_store), + Arc::clone(&wal_store), ) .await .unwrap(); @@ -383,12 +389,13 @@ mod tests { #[tokio::test] async fn should_replay_zero_byte_wal_fence() { - let table_store = test_table_store(); - table_store.write_wal_fence(1).await.unwrap(); + let (table_store, wal_store) = test_stores(); + wal_store.write_wal_fence(1).await.unwrap(); let mut replay_iter = WalReplayIterator::all_wal_ids( &ManifestCore::new(), WalReplayOptions::default(), Arc::clone(&table_store), + Arc::clone(&wal_store), ) .await .unwrap(); @@ -406,22 +413,20 @@ mod tests { #[tokio::test] async fn should_replay_zero_byte_wal_fence_before_real_wal() { - let table_store = test_table_store(); - table_store.write_wal_fence(1).await.unwrap(); + let (table_store, wal_store) = test_stores(); + wal_store.write_wal_fence(1).await.unwrap(); let row = RowEntry::new_value(b"key", b"value", 1); - let mut builder = table_store.wal_table_builder(); + let mut builder = wal_store.table_builder(); builder.add(row.clone()).await.unwrap(); let encoded_sst = builder.build().await.unwrap(); - table_store - .write_sst(&SsTableId::Wal(2), &encoded_sst) - .await - .unwrap(); + wal_store.write_sst(2, &encoded_sst).await.unwrap(); let mut replay_iter = WalReplayIterator::all_wal_ids( &ManifestCore::new(), WalReplayOptions::default(), Arc::clone(&table_store), + Arc::clone(&wal_store), ) .await .unwrap(); @@ -439,10 +444,10 @@ mod tests { #[tokio::test] async fn should_replay_all_entries() { - let table_store = test_table_store(); + let (table_store, wal_store) = test_stores(); let mut rng = rng::new_test_rng(None); let entries = sample::table(&mut rng, 1000, 10); - let next_wal_id = write_wals(&entries, 1, &mut rng, 200, Arc::clone(&table_store)) + let next_wal_id = write_wals(&entries, 1, &mut rng, 200, Arc::clone(&wal_store)) .await .unwrap(); @@ -450,6 +455,7 @@ mod tests { &ManifestCore::new(), WalReplayOptions::default(), Arc::clone(&table_store), + Arc::clone(&wal_store), ) .await .unwrap(); @@ -472,11 +478,11 @@ mod tests { #[tokio::test] async fn should_enforce_max_memtable_bytes() { - let table_store = test_table_store(); + let (table_store, wal_store) = test_stores(); let mut rng = rng::new_test_rng(None); let num_entries = 5000; let entries = sample::table(&mut rng, num_entries, 10); - let next_wal_id = write_wals(&entries, 1, &mut rng, 200, Arc::clone(&table_store)) + let next_wal_id = write_wals(&entries, 1, &mut rng, 200, Arc::clone(&wal_store)) .await .unwrap(); @@ -488,6 +494,7 @@ mod tests { ..WalReplayOptions::default() }, Arc::clone(&table_store), + Arc::clone(&wal_store), ) .await .unwrap(); @@ -529,7 +536,7 @@ mod tests { #[tokio::test] async fn should_apply_max_memtable_bytes_at_wal_boundaries() { - let table_store = test_table_store(); + let (table_store, wal_store) = test_stores(); let wal_entries = [ vec![RowEntry::new_value(b"key_001", &[b'x'; 128], 1)], vec![RowEntry::new_value(b"key_002", &[b'x'; 128], 2)], @@ -540,13 +547,13 @@ mod tests { table_store.estimate_encoded_size_compacted(1, single_row_size) + 1; for (wal_id, entries) in wal_entries.into_iter().enumerate() { - let mut builder = table_store.wal_table_builder(); + let mut builder = wal_store.table_builder(); for entry in entries { builder.add(entry).await.unwrap(); } let encoded_sst = builder.build().await.unwrap(); - table_store - .write_sst(&SsTableId::Wal(wal_id as u64 + 1), &encoded_sst) + wal_store + .write_sst(wal_id as u64 + 1, &encoded_sst) .await .unwrap(); } @@ -558,6 +565,7 @@ mod tests { ..WalReplayOptions::default() }, Arc::clone(&table_store), + Arc::clone(&wal_store), ) .await .unwrap(); @@ -589,7 +597,7 @@ mod tests { #[tokio::test] async fn should_not_split_one_commit_seq_across_replayed_memtables() { - let table_store = test_table_store(); + let (table_store, wal_store) = test_stores(); let commit_seq = 42; // Simulate one committed write batch. Every row gets the same commit @@ -607,15 +615,12 @@ mod tests { table_store.estimate_encoded_size_compacted(1, entries[0].estimated_size()); // Use the real WAL SST builder so the fixture matches WAL flushes. - let mut builder = table_store.wal_table_builder(); + let mut builder = wal_store.table_builder(); for entry in entries { builder.add(entry).await.unwrap(); } let encoded_sst = builder.build().await.unwrap(); - table_store - .write_sst(&SsTableId::Wal(1), &encoded_sst) - .await - .unwrap(); + wal_store.write_sst(1, &encoded_sst).await.unwrap(); // Replay the single WAL SST into in-memory tables. If the replay code // can split a single commit sequence, it will do so here. @@ -626,6 +631,7 @@ mod tests { ..WalReplayOptions::default() }, Arc::clone(&table_store), + Arc::clone(&wal_store), ) .await .unwrap(); @@ -648,7 +654,7 @@ mod tests { #[tokio::test] async fn should_replay_memtables_in_sequence_order() { - let table_store = test_table_store(); + let (table_store, wal_store) = test_stores(); // Write one WAL with entries whose sequence numbers do not match key // order. Replay must not expose a later memtable whose sequence range @@ -666,15 +672,12 @@ mod tests { // Use the real WAL SST builder so replay sees the same entry order as a // flushed WAL. - let mut builder = table_store.wal_table_builder(); + let mut builder = wal_store.table_builder(); for entry in entries { builder.add(entry).await.unwrap(); } let encoded_sst = builder.build().await.unwrap(); - table_store - .write_sst(&SsTableId::Wal(1), &encoded_sst) - .await - .unwrap(); + wal_store.write_sst(1, &encoded_sst).await.unwrap(); // Replay the single WAL SST into in-memory tables. let mut replay_iter = WalReplayIterator::all_wal_ids( @@ -684,6 +687,7 @@ mod tests { ..WalReplayOptions::default() }, Arc::clone(&table_store), + Arc::clone(&wal_store), ) .await .unwrap(); @@ -708,7 +712,7 @@ mod tests { #[tokio::test] async fn should_only_replay_wals_after_last_l0_flushed_wal_id() { - let table_store = test_table_store(); + let (table_store, wal_store) = test_stores(); let mut rng = rng::new_test_rng(None); let compacted_entries = sample::table(&mut rng, 1000, 10); let mut next_wal_id = 1; @@ -718,7 +722,7 @@ mod tests { next_wal_id, &mut rng, 200, - Arc::clone(&table_store), + Arc::clone(&wal_store), ) .await .unwrap(); @@ -730,7 +734,7 @@ mod tests { next_wal_id, &mut rng, 200, - Arc::clone(&table_store), + Arc::clone(&wal_store), ) .await .unwrap(); @@ -743,6 +747,7 @@ mod tests { &db_state, WalReplayOptions::default(), Arc::clone(&table_store), + Arc::clone(&wal_store), ) .await .unwrap(); @@ -765,10 +770,10 @@ mod tests { #[tokio::test] async fn should_replay_wals_after_min_seq() { - let table_store = test_table_store(); + let (table_store, wal_store) = test_stores(); let mut rng = rng::new_test_rng(None); let entries = sample::table(&mut rng, 1000, 10); - let next_wal_id = write_wals(&entries, 1, &mut rng, 200, Arc::clone(&table_store)) + let next_wal_id = write_wals(&entries, 1, &mut rng, 200, Arc::clone(&wal_store)) .await .unwrap(); @@ -782,6 +787,7 @@ mod tests { &db_state, WalReplayOptions::default(), Arc::clone(&table_store), + Arc::clone(&wal_store), ) .await .unwrap(); @@ -804,16 +810,27 @@ mod tests { } fn test_table_store() -> Arc { + test_stores().0 + } + + fn test_stores() -> (Arc, Arc) { let object_store: Arc = Arc::new(InMemory::new()); let path = Path::from("/tmp/test_kv_store"); - Arc::new(TableStore::new( - ObjectStores::new(object_store.clone(), None), + let table_store = Arc::new(TableStore::new( + ObjectStores::new(Arc::clone(&object_store), None), SsTableFormat::default(), - path, + path.clone(), None, TableStoreKind::Main, BlockCachePolicy::default(), - )) + )); + let wal_store = Arc::new(WalTableStore::new( + object_store, + SsTableFormat::default(), + path, + TableStoreKind::Main, + )); + (table_store, wal_store) } /// Write a sequence of WALs with a random (bounded) number of entries. @@ -823,7 +840,7 @@ mod tests { next_wal_id: u64, rng: &mut TestRng, max_wal_entries: usize, - table_store: Arc, + wal_store: Arc, ) -> Result { let mut iter = entries.iter(); let mut next_seq = 1; @@ -840,7 +857,7 @@ mod tests { next_seq, &mut iter, wal_entries, - Arc::clone(&table_store), + Arc::clone(&wal_store), ) .await?; next_wal_id += 1; @@ -851,11 +868,11 @@ mod tests { async fn write_empty_wal( wal_id: u64, - table_store: Arc, + wal_store: Arc, ) -> Result<(), SlateDBError> { let empty_entries = BTreeMap::new(); let mut empty_iter = empty_entries.iter(); - let _ = write_wal(wal_id, 0, &mut empty_iter, 0, table_store).await?; + let _ = write_wal(wal_id, 0, &mut empty_iter, 0, wal_store).await?; Ok(()) } @@ -864,21 +881,22 @@ mod tests { next_seq: u64, entries: &mut Iter<'_, Bytes, Bytes>, max_entries: usize, - table_store: Arc, + wal_store: Arc, ) -> Result { - let mut writer = table_store.table_writer(SsTableId::Wal(wal_id)); + let mut builder = wal_store.table_builder(); let mut next_seq = next_seq; let end_seq = next_seq + (max_entries as u64); while next_seq < end_seq { let Some((key, value)) = entries.next() else { break; }; - writer + builder .add(RowEntry::new_value(key, value, next_seq)) .await?; next_seq += 1; } - writer.close().await?; + let encoded_sst = builder.build().await?; + wal_store.write_sst(wal_id, &encoded_sst).await?; Ok(next_seq) } } From 4919857e75b30f63ce2005c8d4a64786de6c553d Mon Sep 17 00:00:00 2001 From: Vaibhav Srivastava Date: Sat, 29 Aug 2026 20:29:40 +0530 Subject: [PATCH 48/65] docs: fix typo wihtout -> without (#2057) --- slatedb/src/retrying_object_store.rs | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/slatedb/src/retrying_object_store.rs b/slatedb/src/retrying_object_store.rs index 9a9f71b389..ca553ed94c 100644 --- a/slatedb/src/retrying_object_store.rs +++ b/slatedb/src/retrying_object_store.rs @@ -266,7 +266,7 @@ impl ObjectStore for RetryingObjectStore { if options_range.is_none() { // No range requested — don't buffer the body. The buffer size - // can't be validated wihtout buffering. + // can't be validated without buffering. return Ok(result); } From 3fb9e8abab0c9f5833f0c154140ceef009fea02a Mon Sep 17 00:00:00 2001 From: criccomini Date: Mon, 31 Aug 2026 14:58:08 +0000 Subject: [PATCH 49/65] Bump version to 0.16.0 --- Cargo.lock | 16 ++++++++-------- Cargo.toml | 8 ++++---- bindings/java/gradle.properties | 2 +- bindings/node/package.json | 2 +- 4 files changed, 14 insertions(+), 14 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index 53af1e2132..c027072970 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -914,7 +914,7 @@ dependencies = [ [[package]] name = "examples" -version = "0.15.0" +version = "0.16.0" dependencies = [ "anyhow", "object_store", @@ -3211,7 +3211,7 @@ checksum = "0c790de23124f9ab44544d7ac05d60440adc586479ce501c1d6d7da3cd8c9cf5" [[package]] name = "slatedb" -version = "0.15.0" +version = "0.16.0" dependencies = [ "async-channel", "async-trait", @@ -3271,7 +3271,7 @@ dependencies = [ [[package]] name = "slatedb-bencher" -version = "0.15.0" +version = "0.16.0" dependencies = [ "bytes", "chrono", @@ -3289,7 +3289,7 @@ dependencies = [ [[package]] name = "slatedb-cli" -version = "0.15.0" +version = "0.16.0" dependencies = [ "chrono", "clap", @@ -3309,7 +3309,7 @@ dependencies = [ [[package]] name = "slatedb-common" -version = "0.15.0" +version = "0.16.0" dependencies = [ "chrono", "log", @@ -3323,7 +3323,7 @@ dependencies = [ [[package]] name = "slatedb-dst" -version = "0.15.0" +version = "0.16.0" dependencies = [ "async-trait", "bytes", @@ -3348,7 +3348,7 @@ dependencies = [ [[package]] name = "slatedb-txn-obj" -version = "0.15.0" +version = "0.16.0" dependencies = [ "async-trait", "bytes", @@ -3365,7 +3365,7 @@ dependencies = [ [[package]] name = "slatedb-uniffi" -version = "0.15.0" +version = "0.16.0" dependencies = [ "chrono", "figment", diff --git a/Cargo.toml b/Cargo.toml index b4a888816b..6d0a7498c2 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -12,7 +12,7 @@ members = [ ] [workspace.package] -version = "0.15.0" +version = "0.16.0" edition = "2021" repository = "https://github.com/slatedb/slatedb" license = "Apache-2.0" @@ -69,9 +69,9 @@ serde = "1.0" serde_json = "1.0.142" siphasher = "1" smallvec = "1.15.1" -slatedb = { path = "slatedb", version = "0.15.0" } -slatedb-common = { path = "slatedb-common", version = "0.15.0" } -slatedb-txn-obj = { path = "slatedb-txn-obj", version = "0.15.0" } +slatedb = { path = "slatedb", version = "0.16.0" } +slatedb-common = { path = "slatedb-common", version = "0.16.0" } +slatedb-txn-obj = { path = "slatedb-txn-obj", version = "0.16.0" } snap = "1.1.1" sysinfo = "0.35.2" thiserror = "1.0.63" diff --git a/bindings/java/gradle.properties b/bindings/java/gradle.properties index dfb204fc38..baf14210e5 100644 --- a/bindings/java/gradle.properties +++ b/bindings/java/gradle.properties @@ -1 +1 @@ -version=0.15.0-SNAPSHOT +version=0.16.0-SNAPSHOT diff --git a/bindings/node/package.json b/bindings/node/package.json index 91bc2b891c..a35401c454 100644 --- a/bindings/node/package.json +++ b/bindings/node/package.json @@ -1,6 +1,6 @@ { "name": "@slatedb/uniffi", - "version": "0.15.0", + "version": "0.16.0", "description": "Node.js bindings for SlateDB generated from UniFFI and packaged with native libraries.", "license": "Apache-2.0", "type": "module", From 20e4165e5353f729b701f3a7532262309d8797ed Mon Sep 17 00:00:00 2001 From: xav-db Date: Thu, 16 Jul 2026 14:16:25 +0100 Subject: [PATCH 50/65] Add request-scoped query storage metrics --- .../src/cached_object_store/object_store.rs | 16 +- slatedb/src/db_cache/mod.rs | 19 ++- slatedb/src/instrumented_object_store.rs | 6 + slatedb/src/lib.rs | 5 + slatedb/src/query_metrics.rs | 151 ++++++++++++++++++ 5 files changed, 184 insertions(+), 13 deletions(-) create mode 100644 slatedb/src/query_metrics.rs diff --git a/slatedb/src/cached_object_store/object_store.rs b/slatedb/src/cached_object_store/object_store.rs index 8ffc9f82e9..1cf2fbdabb 100644 --- a/slatedb/src/cached_object_store/object_store.rs +++ b/slatedb/src/cached_object_store/object_store.rs @@ -7,6 +7,7 @@ use crate::cached_object_store::storage_fs::FsCacheStorage; use crate::cached_object_store::LocalCacheEntry; use crate::config::ObjectStoreCacheOptions; use crate::object_store_tag::ObjectStoreCallTag; +use crate::query_metrics::{self, QueryCacheKind}; use bytes::{Bytes, BytesMut}; use futures::{future::BoxFuture, stream, stream::BoxStream, StreamExt}; use object_store::{path::Path, GetOptions, GetResult, ObjectMeta, ObjectStore, ObjectStoreExt}; @@ -310,12 +311,17 @@ impl CachedObjectStore { let location = location.clone(); async move { this.stats.object_store_cache_part_access.increment(1); - let (bytes, part_source) = this + let result = this .read_part(&location, part_id, range_in_part, force_refresh) - .await?; - if head_source == ReadResultSource::Disk - && part_source == ReadResultSource::Disk - { + .await; + let Ok((bytes, part_source)) = result else { + query_metrics::record_cache_access(QueryCacheKind::Object, false); + return result.map(|(bytes, _)| bytes); + }; + let hit = head_source == ReadResultSource::Disk + && part_source == ReadResultSource::Disk; + query_metrics::record_cache_access(QueryCacheKind::Object, hit); + if hit { this.stats.object_store_cache_part_hits.increment(1); } Ok::(bytes) diff --git a/slatedb/src/db_cache/mod.rs b/slatedb/src/db_cache/mod.rs index e4662e19bd..78f0cca08e 100644 --- a/slatedb/src/db_cache/mod.rs +++ b/slatedb/src/db_cache/mod.rs @@ -27,6 +27,7 @@ use crate::db_state::SsTableId; use crate::filter_policy::NamedFilter; use crate::flatbuffer_types::SsTableIndexOwned; use crate::format::block::Block; +use crate::query_metrics::{self, QueryCacheKind}; use crate::sst_stats::SstStats; use slatedb_common::clock::SystemClock; use slatedb_common::metrics::MetricsRecorderHelper; @@ -683,6 +684,7 @@ impl DbCacheWrapper { } fn record_hit(&self, block_type: &str) { + query_metrics::record_cache_access(QueryCacheKind::Block, true); match block_type { "block" => self.stats.data_block_hit.increment(1), "index" => self.stats.index_hit.increment(1), @@ -693,6 +695,7 @@ impl DbCacheWrapper { } fn record_miss(&self, block_type: &str) { + query_metrics::record_cache_access(QueryCacheKind::Block, false); match block_type { "block" => self.stats.data_block_miss.increment(1), "index" => self.stats.index_miss.increment(1), @@ -744,9 +747,9 @@ impl DbCache for DbCacheWrapper { } }; if entry.is_some() { - self.stats.data_block_hit.increment(1); + self.record_hit("block"); } else { - self.stats.data_block_miss.increment(1); + self.record_miss("block"); } Ok(entry) } @@ -761,9 +764,9 @@ impl DbCache for DbCacheWrapper { } }; if entry.is_some() { - self.stats.index_hit.increment(1); + self.record_hit("index"); } else { - self.stats.index_miss.increment(1); + self.record_miss("index"); } Ok(entry) } @@ -778,9 +781,9 @@ impl DbCache for DbCacheWrapper { } }; if entry.is_some() { - self.stats.filter_hit.increment(1); + self.record_hit("filter"); } else { - self.stats.filter_miss.increment(1); + self.record_miss("filter"); } Ok(entry) } @@ -795,9 +798,9 @@ impl DbCache for DbCacheWrapper { } }; if entry.is_some() { - self.stats.stats_hit.increment(1); + self.record_hit("stats"); } else { - self.stats.stats_miss.increment(1); + self.record_miss("stats"); } Ok(entry) } diff --git a/slatedb/src/instrumented_object_store.rs b/slatedb/src/instrumented_object_store.rs index 822111e47f..e87d632b30 100644 --- a/slatedb/src/instrumented_object_store.rs +++ b/slatedb/src/instrumented_object_store.rs @@ -42,6 +42,7 @@ use object_store::{ use slatedb_common::metrics::MetricsRecorderHelper; use crate::object_stores::ObjectStoreType; +use crate::query_metrics; /// Which SlateDB component is issuing object store requests. /// @@ -155,6 +156,7 @@ impl ObjectStore for InstrumentedObjectStore { location: &Path, options: GetOptions, ) -> object_store::Result { + query_metrics::record_object_storage_read(); let metric = if options.head { &self.stats.head } else if options.range.is_some() { @@ -173,6 +175,7 @@ impl ObjectStore for InstrumentedObjectStore { location: &Path, ranges: &[Range], ) -> object_store::Result> { + query_metrics::record_object_storage_read(); let start = Instant::now(); let result = self.inner.get_ranges(location, ranges).await; self.stats @@ -233,6 +236,7 @@ impl ObjectStore for InstrumentedObjectStore { } fn list(&self, prefix: Option<&Path>) -> BoxStream<'static, object_store::Result> { + query_metrics::record_object_storage_read(); self.stats.list.increment(1); self.inner.list(prefix) } @@ -242,11 +246,13 @@ impl ObjectStore for InstrumentedObjectStore { prefix: Option<&Path>, offset: &Path, ) -> BoxStream<'static, object_store::Result> { + query_metrics::record_object_storage_read(); self.stats.list_with_offset.increment(1); self.inner.list_with_offset(prefix, offset) } async fn list_with_delimiter(&self, prefix: Option<&Path>) -> object_store::Result { + query_metrics::record_object_storage_read(); let start = Instant::now(); let result = self.inner.list_with_delimiter(prefix).await; self.stats diff --git a/slatedb/src/lib.rs b/slatedb/src/lib.rs index 3d86763299..875ea32240 100644 --- a/slatedb/src/lib.rs +++ b/slatedb/src/lib.rs @@ -69,6 +69,10 @@ pub use merge_operator::{MergeOperator, MergeOperatorError}; pub use ops::{DbCacheManagerOps, DbMetadataOps, DbReadOps, DbTransactionOps, DbWriteOps}; pub use paths::PathResolver; pub use prefix_extractor::{PrefixExtractor, PrefixTarget}; +pub use query_metrics::{ + scope_query_metrics, QueryCacheKind, QueryCacheStatistics, QueryMetricsObserver, + QueryMetricsSnapshot, +}; pub use slatedb_common::{DbRand, IdentifiedObjectMetadata, ObjectMetadata}; #[cfg(test)] pub use sst_builder::BlockFormat; @@ -93,6 +97,7 @@ pub mod db_stats; pub mod manifest; pub mod object_store_tag; pub mod prefix_extractor; +pub mod query_metrics; pub mod seq_tracker; pub mod size_tiered_compaction; pub mod wal; diff --git a/slatedb/src/query_metrics.rs b/slatedb/src/query_metrics.rs new file mode 100644 index 0000000000..8ab2d64f03 --- /dev/null +++ b/slatedb/src/query_metrics.rs @@ -0,0 +1,151 @@ +//! Request-scoped storage metrics for foreground query execution. +//! +//! The observer is installed with [`scope_query_metrics`]. Storage paths record +//! against the current task scope without reading or subtracting process-wide +//! metrics, so concurrent queries cannot contaminate each other's counters. + +use std::future::Future; +use std::sync::atomic::{AtomicU64, Ordering}; +use std::sync::Arc; + +tokio::task_local! { + static QUERY_METRICS: QueryMetricsObserver; +} + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum QueryCacheKind { + Block, + Object, +} + +#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] +pub struct QueryCacheStatistics { + pub hits: u64, + pub misses: u64, +} + +#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] +pub struct QueryMetricsSnapshot { + pub block_cache: QueryCacheStatistics, + pub object_cache: QueryCacheStatistics, + pub object_storage_reads: u64, +} + +#[derive(Clone, Debug, Default)] +pub struct QueryMetricsObserver { + inner: Arc, +} + +#[derive(Debug, Default)] +struct QueryMetricsInner { + block_cache_hits: AtomicU64, + block_cache_misses: AtomicU64, + object_cache_hits: AtomicU64, + object_cache_misses: AtomicU64, + object_storage_reads: AtomicU64, +} + +impl QueryMetricsObserver { + #[must_use] + pub fn snapshot(&self) -> QueryMetricsSnapshot { + QueryMetricsSnapshot { + block_cache: QueryCacheStatistics { + hits: self.inner.block_cache_hits.load(Ordering::Relaxed), + misses: self.inner.block_cache_misses.load(Ordering::Relaxed), + }, + object_cache: QueryCacheStatistics { + hits: self.inner.object_cache_hits.load(Ordering::Relaxed), + misses: self.inner.object_cache_misses.load(Ordering::Relaxed), + }, + object_storage_reads: self.inner.object_storage_reads.load(Ordering::Relaxed), + } + } + + fn record_cache(&self, kind: QueryCacheKind, hit: bool) { + let counter = match (kind, hit) { + (QueryCacheKind::Block, true) => &self.inner.block_cache_hits, + (QueryCacheKind::Block, false) => &self.inner.block_cache_misses, + (QueryCacheKind::Object, true) => &self.inner.object_cache_hits, + (QueryCacheKind::Object, false) => &self.inner.object_cache_misses, + }; + counter.fetch_add(1, Ordering::Relaxed); + } + + fn record_object_storage_read(&self) { + self.inner + .object_storage_reads + .fetch_add(1, Ordering::Relaxed); + } +} + +/// Runs one foreground query with an isolated storage observer. +/// +/// ``` +/// # tokio_test::block_on(async { +/// use slatedb::{QueryMetricsObserver, scope_query_metrics}; +/// +/// let observer = QueryMetricsObserver::default(); +/// scope_query_metrics(observer.clone(), async {}).await; +/// assert_eq!(observer.snapshot().object_storage_reads, 0); +/// # }); +/// ``` +pub async fn scope_query_metrics(observer: QueryMetricsObserver, future: F) -> F::Output +where + F: Future, +{ + QUERY_METRICS.scope(observer, future).await +} + +pub(crate) fn record_cache_access(kind: QueryCacheKind, hit: bool) { + let _ = QUERY_METRICS.try_with(|observer| observer.record_cache(kind, hit)); +} + +pub(crate) fn record_object_storage_read() { + let _ = QUERY_METRICS.try_with(QueryMetricsObserver::record_object_storage_read); +} + +#[cfg(test)] +mod tests { + use super::*; + + #[tokio::test] + async fn concurrent_scopes_do_not_share_counters() { + let first = QueryMetricsObserver::default(); + let second = QueryMetricsObserver::default(); + + tokio::join!( + scope_query_metrics(first.clone(), async { + record_cache_access(QueryCacheKind::Block, true); + tokio::task::yield_now().await; + record_object_storage_read(); + }), + scope_query_metrics(second.clone(), async { + record_cache_access(QueryCacheKind::Object, false); + tokio::task::yield_now().await; + record_cache_access(QueryCacheKind::Object, false); + }), + ); + + assert_eq!( + first.snapshot(), + QueryMetricsSnapshot { + block_cache: QueryCacheStatistics { hits: 1, misses: 0 }, + object_storage_reads: 1, + ..QueryMetricsSnapshot::default() + } + ); + assert_eq!( + second.snapshot(), + QueryMetricsSnapshot { + object_cache: QueryCacheStatistics { hits: 0, misses: 2 }, + ..QueryMetricsSnapshot::default() + } + ); + } + + #[tokio::test] + async fn records_outside_scope_are_ignored() { + record_cache_access(QueryCacheKind::Block, false); + record_object_storage_read(); + } +} From 5c2ad293862934683e15d463d8a51f7437d33bbe Mon Sep 17 00:00:00 2001 From: Matthew Sanetra <41018997+matthewsanetra@users.noreply.github.com> Date: Thu, 16 Jul 2026 15:55:41 +0100 Subject: [PATCH 51/65] Add snapshot isolation to `DbReader` --- slatedb-dst/src/actors/bank/auditor.rs | 107 +- slatedb-dst/tests/bank.rs | 12 +- slatedb/Cargo.toml | 10 + slatedb/benches/db_reader_memory_scaling.rs | 501 ++++++ slatedb/benches/db_reader_scaling.rs | 1250 ++++++++++++++ slatedb/src/db_cache/mod.rs | 68 +- slatedb/src/db_iter.rs | 45 + slatedb/src/db_reader.rs | 1714 +++++++++++++++++-- slatedb/src/db_snapshot.rs | 119 +- slatedb/src/db_state.rs | 6 +- slatedb/src/db_stats.rs | 26 + slatedb/src/error.rs | 12 + slatedb/src/manifest/store.rs | 252 ++- slatedb/src/reader.rs | 12 +- 14 files changed, 3929 insertions(+), 205 deletions(-) create mode 100644 slatedb/benches/db_reader_memory_scaling.rs create mode 100644 slatedb/benches/db_reader_scaling.rs diff --git a/slatedb-dst/src/actors/bank/auditor.rs b/slatedb-dst/src/actors/bank/auditor.rs index cdb3b93498..99d7e14ad0 100644 --- a/slatedb-dst/src/actors/bank/auditor.rs +++ b/slatedb-dst/src/actors/bank/auditor.rs @@ -1,8 +1,11 @@ +use std::collections::{HashMap, VecDeque}; +use std::sync::{Arc, Mutex}; use std::time::Duration; use async_trait::async_trait; use rand::RngCore; use slatedb::config::DbReaderOptions; +use slatedb::db_cache::{CachedEntry, CachedKey, DbCache}; use slatedb::{DbReadOps, DbReader, Error}; use tracing::{info, instrument}; @@ -19,6 +22,8 @@ pub enum BankAuditView { Snapshot, /// Audit a long-lived read-only `DbReader`. Reader { options: DbReaderOptions }, + /// Audit an O(1) snapshot captured from a long-lived read-only `DbReader`. + ReaderSnapshot { options: DbReaderOptions }, } impl BankAuditView { @@ -27,6 +32,7 @@ impl BankAuditView { Self::Regular => "db", Self::Snapshot => "db_snapshot", Self::Reader { .. } => "db_reader", + Self::ReaderSnapshot { .. } => "db_reader_snapshot", } } } @@ -89,6 +95,21 @@ impl Actor for AuditorActor { .expect("bank reader auditor should have opened a reader"); audit_bank_view(reader, &self.bank, self.step).await?; } + BankAuditView::ReaderSnapshot { options } => { + if self.reader.is_none() { + self.reader = Some(open_bank_reader(ctx, options).await?); + } + let reader = self + .reader + .as_ref() + .expect("bank reader snapshot auditor should have opened a reader"); + let snapshot = reader.snapshot().await?; + // Make the capture-to-read window explicit so transfer, flush, + // compaction, GC, and fencing actors can advance after this + // snapshot has fixed its manifest generation and sequence. + tokio::task::yield_now().await; + audit_bank_view(snapshot.as_ref(), &self.bank, self.step).await?; + } }; self.step += 1; @@ -121,7 +142,11 @@ async fn open_bank_reader(ctx: &ActorCtx, options: DbReaderOptions) -> Result Result, +} + +#[derive(Default)] +struct DeterministicDbCacheInner { + entries: HashMap, + insertion_order: VecDeque, + size: usize, +} + +impl DeterministicDbCache { + fn new(capacity: usize) -> Self { + Self { + capacity, + inner: Mutex::new(DeterministicDbCacheInner::default()), + } + } + + fn get(&self, key: &CachedKey) -> Option { + self.inner.lock().unwrap().entries.get(key).cloned() + } +} + +#[async_trait] +impl DbCache for DeterministicDbCache { + async fn get_block(&self, key: &CachedKey) -> Result, Error> { + Ok(self.get(key)) + } + + async fn get_index(&self, key: &CachedKey) -> Result, Error> { + Ok(self.get(key)) + } + + async fn get_filter(&self, key: &CachedKey) -> Result, Error> { + Ok(self.get(key)) + } + + async fn get_stats(&self, key: &CachedKey) -> Result, Error> { + Ok(self.get(key)) + } + + async fn insert(&self, key: CachedKey, value: CachedEntry) { + let value_size = value.size(); + if value_size > self.capacity { + return; + } + + let mut inner = self.inner.lock().unwrap(); + if let Some(previous) = inner.entries.insert(key.clone(), value) { + inner.size = inner.size.saturating_sub(previous.size()) + value_size; + } else { + inner.size += value_size; + inner.insertion_order.push_back(key); + } + + while inner.size > self.capacity { + let Some(oldest) = inner.insertion_order.pop_front() else { + break; + }; + if let Some(evicted) = inner.entries.remove(&oldest) { + inner.size = inner.size.saturating_sub(evicted.size()); + } + } + } + + async fn remove(&self, key: &CachedKey) { + let mut inner = self.inner.lock().unwrap(); + if let Some(removed) = inner.entries.remove(key) { + inner.size = inner.size.saturating_sub(removed.size()); + } + inner.insertion_order.retain(|queued| queued != key); + } + + fn entry_count(&self) -> u64 { + self.inner.lock().unwrap().entries.len() as u64 + } +} + async fn audit_bank_view(reader: &R, bank: &BankAccounts, step: u64) -> Result<(), Error> where R: DbReadOps + Sync, diff --git a/slatedb-dst/tests/bank.rs b/slatedb-dst/tests/bank.rs index 5462536daa..03c22e6b48 100644 --- a/slatedb-dst/tests/bank.rs +++ b/slatedb-dst/tests/bank.rs @@ -152,9 +152,19 @@ fn run_bank(seed: u64, shutdown_at_ms: i64) -> Result<(), Box *mut u8 { + // SAFETY: Delegates the exact layout to the system allocator. + let pointer = unsafe { System.alloc(layout) }; + if !pointer.is_null() { + record_allocation(layout.size()); + } + pointer + } + + unsafe fn alloc_zeroed(&self, layout: Layout) -> *mut u8 { + // SAFETY: Delegates the exact layout to the system allocator. + let pointer = unsafe { System.alloc_zeroed(layout) }; + if !pointer.is_null() { + record_allocation(layout.size()); + } + pointer + } + + unsafe fn dealloc(&self, pointer: *mut u8, layout: Layout) { + LIVE_BYTES.fetch_sub(layout.size(), Ordering::Relaxed); + // SAFETY: `pointer` was allocated with this allocator and `layout` is unchanged. + unsafe { System.dealloc(pointer, layout) }; + } + + unsafe fn realloc(&self, pointer: *mut u8, old: Layout, new_size: usize) -> *mut u8 { + // SAFETY: Delegates the original pointer/layout and requested size unchanged. + let new_pointer = unsafe { System.realloc(pointer, old, new_size) }; + if !new_pointer.is_null() { + ALLOCATED_BYTES.fetch_add(new_size, Ordering::Relaxed); + ALLOCATION_CALLS.fetch_add(1, Ordering::Relaxed); + let live = if new_size >= old.size() { + LIVE_BYTES.fetch_add(new_size - old.size(), Ordering::Relaxed) + new_size + - old.size() + } else { + LIVE_BYTES.fetch_sub(old.size() - new_size, Ordering::Relaxed) + - (old.size() - new_size) + }; + PEAK_LIVE_BYTES.fetch_max(live, Ordering::Relaxed); + } + new_pointer + } +} + +#[global_allocator] +static GLOBAL_ALLOCATOR: TrackingAllocator = TrackingAllocator; + +#[derive(Clone, Copy, Debug)] +struct MemoryWindow { + live_before: usize, + allocated_before: usize, + calls_before: usize, +} + +#[derive(Clone, Copy, Debug)] +struct MemorySample { + retained_bytes: i64, + peak_extra_bytes: usize, + allocated_bytes: usize, + allocation_calls: usize, +} + +impl MemoryWindow { + fn start() -> Self { + let live_before = LIVE_BYTES.load(Ordering::SeqCst); + PEAK_LIVE_BYTES.store(live_before, Ordering::SeqCst); + Self { + live_before, + allocated_before: ALLOCATED_BYTES.load(Ordering::SeqCst), + calls_before: ALLOCATION_CALLS.load(Ordering::SeqCst), + } + } + + fn finish(self) -> MemorySample { + let live_after = LIVE_BYTES.load(Ordering::SeqCst); + MemorySample { + retained_bytes: live_after as i64 - self.live_before as i64, + peak_extra_bytes: PEAK_LIVE_BYTES + .load(Ordering::SeqCst) + .saturating_sub(self.live_before), + allocated_bytes: ALLOCATED_BYTES + .load(Ordering::SeqCst) + .saturating_sub(self.allocated_before), + allocation_calls: ALLOCATION_CALLS + .load(Ordering::SeqCst) + .saturating_sub(self.calls_before), + } + } +} + +#[derive(Clone, Copy)] +enum Segmentation { + None, + Fixed4, +} + +impl Segmentation { + fn label(self) -> &'static str { + match self { + Self::None => "unsegmented", + Self::Fixed4 => "segmented", + } + } +} + +struct Fixed4Extractor; + +impl PrefixExtractor for Fixed4Extractor { + fn name(&self) -> &str { + "reader-memory-bench-fixed4" + } + + fn prefix_len(&self, target: &PrefixTarget) -> Option { + let len = match target { + PrefixTarget::Point(key) | PrefixTarget::Prefix(key) => key.len(), + }; + (len >= 4).then_some(4) + } +} + +fn writer_settings() -> Settings { + Settings { + flush_interval: None, + compactor_options: None, + garbage_collector_options: None, + l0_sst_size_bytes: 256 * 1024 * 1024, + l0_max_ssts: 16_384, + l0_max_ssts_per_key: 16_384, + ..Settings::default() + } +} + +fn reader_options(max_memtable_bytes: u64) -> DbReaderOptions { + DbReaderOptions { + manifest_poll_interval: POLL_INTERVAL, + checkpoint_lifetime: Duration::from_secs(60), + max_memtable_bytes, + ..DbReaderOptions::default() + } +} + +fn key_for(index: usize, segmentation: Segmentation) -> Bytes { + match segmentation { + Segmentation::None => Bytes::from(format!("key-{index:08}")), + Segmentation::Fixed4 => Bytes::from(format!("{index:04}-key-{index:08}")), + } +} + +async fn open_db( + path: &str, + store: Arc, + clock: Arc, + segmentation: Segmentation, +) -> Db { + let mut builder = Db::builder(path, store) + .with_settings(writer_settings()) + .with_system_clock(clock); + if matches!(segmentation, Segmentation::Fixed4) { + builder = builder.with_segment_extractor(Arc::new(Fixed4Extractor)); + } + builder.build().await.expect("DB open failed") +} + +async fn open_reader( + path: &str, + store: Arc, + clock: Arc, + segmentation: Segmentation, + recorder: Arc, +) -> DbReader { + let mut builder = DbReader::builder(path, store) + .with_options(reader_options(1)) + .with_system_clock(clock) + .with_metrics_recorder(recorder.clone()) + .with_db_cache_disabled(); + if matches!(segmentation, Segmentation::Fixed4) { + builder = builder.with_segment_extractor(Arc::new(Fixed4Extractor)); + } + let reader = builder.build().await.expect("reader open failed"); + wait_for_counter(|| scalar(&recorder, MANIFEST_POLLS), 1).await; + reader +} + +async fn write_wal(db: &Db, index: usize, segmentation: Segmentation) { + db.put_with_options( + &key_for(index, segmentation), + b"value-value-value-value-value", + &PutOptions::default(), + &WriteOptions { + await_durable: false, + ..WriteOptions::default() + }, + ) + .await + .expect("put failed"); + db.flush_with_options(FlushOptions { + flush_type: FlushType::Wal, + }) + .await + .expect("WAL flush failed"); +} + +fn scalar(recorder: &DefaultMetricsRecorder, name: &str) -> u64 { + lookup_metric(recorder, name).unwrap_or(0) as u64 +} + +async fn wait_for_counter(mut current: F, target: u64) +where + F: FnMut() -> u64, +{ + for _ in 0..100_000 { + if current() >= target { + return; + } + tokio::task::yield_now().await; + } + panic!( + "timed out waiting for reader poll: current={}, target={target}", + current() + ); +} + +async fn settle() { + for _ in 0..100 { + tokio::task::yield_now().await; + } +} + +struct ReaderFixture { + db: Db, + reader: Arc, + recorder: Arc, + clock: Arc, + segmentation: Segmentation, + next_key: usize, +} + +impl ReaderFixture { + async fn new(history: usize, segmentation: Segmentation) -> Self { + let path = format!("bench/db-reader-memory/{}", Uuid::new_v4()); + let store = Arc::new(InMemory::new()); + let clock = Arc::new(MockSystemClock::new()); + let db = open_db(&path, Arc::clone(&store), Arc::clone(&clock), segmentation).await; + for index in 0..history { + write_wal(&db, index, segmentation).await; + } + let recorder = Arc::new(DefaultMetricsRecorder::new()); + let reader = Arc::new( + open_reader( + &path, + store, + Arc::clone(&clock), + segmentation, + Arc::clone(&recorder), + ) + .await, + ); + Self { + db, + reader, + recorder, + clock, + segmentation, + next_key: history, + } + } + + async fn append_wals(&mut self, count: usize) { + for _ in 0..count { + write_wal(&self.db, self.next_key, self.segmentation).await; + self.next_key += 1; + } + } + + async fn poll(&self, expected_new_wals: u64) { + let poll_target = scalar(&self.recorder, MANIFEST_POLLS) + 1; + let replay_target = scalar(&self.recorder, REPLAY_SSTS) + expected_new_wals; + self.clock.advance(POLL_INTERVAL).await; + wait_for_counter(|| scalar(&self.recorder, MANIFEST_POLLS), poll_target).await; + if expected_new_wals > 0 { + wait_for_counter(|| scalar(&self.recorder, REPLAY_SSTS), replay_target).await; + } + } + + async fn close(self) { + self.reader.close().await.expect("reader close failed"); + self.db.close().await.expect("DB close failed"); + } +} + +fn percentile_i64(samples: &[MemorySample], field: impl Fn(&MemorySample) -> i64) -> i64 { + let mut values = samples.iter().map(field).collect::>(); + values.sort_unstable(); + values[(values.len() - 1) / 2] +} + +fn percentile_usize(samples: &[MemorySample], field: impl Fn(&MemorySample) -> usize) -> usize { + let mut values = samples.iter().map(field).collect::>(); + values.sort_unstable(); + values[(values.len() - 1) / 2] +} + +fn print_samples( + suite: &str, + case: &str, + history: usize, + delta: usize, + samples: &[MemorySample], + notes: &str, +) { + println!( + "MEMORY\t{suite}\t{case}\t{history}\t{delta}\t{}\t{}\t{}\t{}\t{}\t{notes}", + samples.len(), + percentile_i64(samples, |sample| sample.retained_bytes), + percentile_usize(samples, |sample| sample.peak_extra_bytes), + percentile_usize(samples, |sample| sample.allocated_bytes), + percentile_usize(samples, |sample| sample.allocation_calls), + ); +} + +async fn benchmark_incremental_replay_memory(full: bool) { + println!("SECTION\tincremental_replay_memory"); + let histories = if full { + vec![0, 32, 128, 512] + } else { + vec![0, 128] + }; + let deltas = if full { vec![1, 8, 32] } else { vec![1, 8] }; + let reps = if full { 9 } else { 5 }; + for segmentation in [Segmentation::None, Segmentation::Fixed4] { + for &history in &histories { + for &delta in &deltas { + let mut samples = Vec::with_capacity(reps); + for _ in 0..reps { + let mut fixture = ReaderFixture::new(history, segmentation).await; + fixture.append_wals(delta).await; + settle().await; + let window = MemoryWindow::start(); + fixture.poll(delta as u64).await; + settle().await; + samples.push(window.finish()); + fixture.close().await; + } + print_samples( + "replay", + "poll_delta", + history, + delta, + &samples, + segmentation.label(), + ); + } + } + } +} + +async fn benchmark_snapshot_memory(full: bool) { + println!("SECTION\tsnapshot_memory"); + let histories = if full { + vec![0, 128, 512] + } else { + vec![0, 128] + }; + let snapshot_count = if full { 10_000 } else { 2_000 }; + for history in histories { + let fixture = ReaderFixture::new(history, Segmentation::None).await; + let mut snapshots: Vec> = Vec::with_capacity(snapshot_count); + settle().await; + let window = MemoryWindow::start(); + for _ in 0..snapshot_count { + snapshots.push( + fixture + .reader + .snapshot() + .await + .expect("snapshot creation failed"), + ); + } + black_box(&snapshots); + settle().await; + let held = window.finish(); + let live_before_drop = LIVE_BYTES.load(Ordering::SeqCst); + snapshots.clear(); + settle().await; + let released = live_before_drop.saturating_sub(LIVE_BYTES.load(Ordering::SeqCst)); + println!( + "SNAPSHOT\thistory={history}\tcount={snapshot_count}\tretained_bytes={}\tbytes_per_snapshot={:.2}\talloc_calls_per_snapshot={:.2}\treleased_bytes={released}", + held.retained_bytes, + held.retained_bytes as f64 / snapshot_count as f64, + held.allocation_calls as f64 / snapshot_count as f64, + ); + fixture.close().await; + } +} + +async fn benchmark_old_generation_release(full: bool) { + println!("SECTION\told_generation_release"); + let histories = if full { + vec![1, 8, 32, 128, 512] + } else { + vec![8, 128] + }; + let reps = if full { 5 } else { 3 }; + for history in histories { + let mut released_samples = Vec::with_capacity(reps); + for _ in 0..reps { + let fixture = ReaderFixture::new(history, Segmentation::None).await; + let old_snapshot = fixture.reader.snapshot().await.expect("snapshot failed"); + fixture + .db + .flush_with_options(FlushOptions { + flush_type: FlushType::MemTable, + }) + .await + .expect("L0 flush failed"); + fixture.poll(0).await; + settle().await; + let before = LIVE_BYTES.load(Ordering::SeqCst); + drop(old_snapshot); + settle().await; + let after = LIVE_BYTES.load(Ordering::SeqCst); + released_samples.push(before.saturating_sub(after)); + fixture.close().await; + } + released_samples.sort_unstable(); + println!( + "RELEASE\thistory={history}\treps={reps}\treleased_bytes_p50={}", + released_samples[(released_samples.len() - 1) / 2] + ); + } +} + +async fn run() { + let full = std::env::var_os("SLATEDB_READER_BENCH_FULL").is_some(); + println!( + "CONFIG\tprofile={}\tfull={full}\tallocator=system-counting", + if cfg!(debug_assertions) { + "debug" + } else { + "release" + } + ); + println!("HEADER\tsuite\tcase\thistory\tdelta\treps\tretained_p50_bytes\tpeak_extra_p50_bytes\tallocated_p50_bytes\tallocation_calls_p50\tnotes"); + benchmark_incremental_replay_memory(full).await; + benchmark_snapshot_memory(full).await; + benchmark_old_generation_release(full).await; +} + +fn main() { + let runtime = tokio::runtime::Builder::new_multi_thread() + .enable_all() + .build() + .expect("failed to build Tokio runtime"); + runtime.block_on(run()); +} diff --git a/slatedb/benches/db_reader_scaling.rs b/slatedb/benches/db_reader_scaling.rs new file mode 100644 index 0000000000..278cb4ea06 --- /dev/null +++ b/slatedb/benches/db_reader_scaling.rs @@ -0,0 +1,1250 @@ +//! End-to-end scaling benchmarks for reader-backed snapshots and incremental WAL replay. +//! +//! This target deliberately uses fixed repetitions instead of Criterion's adaptive +//! iteration count. Replay, checkpoint rollover, and generation cleanup mutate state, +//! and their untimed fixture setup is substantially more expensive than the operation +//! under test. Fixed repetitions keep those costs out of the samples without causing +//! Criterion to request millions of fixture rebuilds. +//! +//! Run the standard matrix: +//! cargo bench -p slatedb --bench db_reader_scaling --features test-util +//! +//! Run the extended edge-case matrix: +//! SLATEDB_READER_BENCH_FULL=1 cargo bench -p slatedb --bench db_reader_scaling --features test-util + +use std::future::Future; +use std::hint::black_box; +use std::sync::Arc; +use std::time::{Duration, Instant}; + +use bytes::Bytes; +use object_store::memory::InMemory; +use object_store::path::Path; +use object_store::{ObjectStore, ObjectStoreExt}; +use slatedb::config::{ + CheckpointOptions, CheckpointScope, DbReaderOptions, FlushOptions, FlushType, PutOptions, + Settings, WriteOptions, +}; +use slatedb::instrumented_object_store_stats; +use slatedb::{Db, DbReader, DbSnapshot, PrefixExtractor, PrefixTarget}; +use slatedb_common::metrics::{lookup_metric, lookup_metric_with_labels, DefaultMetricsRecorder}; +use slatedb_common::{MockSystemClock, SystemClock}; +use tokio::sync::Barrier; +use uuid::Uuid; + +const POLL_INTERVAL: Duration = Duration::from_millis(10); +const REPLAY_SSTS: &str = "slatedb.db.reader_wal_replay_ssts"; +const MANIFEST_POLLS: &str = "slatedb.db.reader_manifest_polls"; + +#[derive(Clone, Copy)] +enum Segmentation { + None, + Fixed4, +} + +impl Segmentation { + fn label(self) -> &'static str { + match self { + Self::None => "unsegmented", + Self::Fixed4 => "segmented", + } + } +} + +struct Fixed4Extractor; + +impl PrefixExtractor for Fixed4Extractor { + fn name(&self) -> &str { + "reader-bench-fixed4" + } + + fn prefix_len(&self, target: &PrefixTarget) -> Option { + let len = match target { + PrefixTarget::Point(key) | PrefixTarget::Prefix(key) => key.len(), + }; + (len >= 4).then_some(4) + } +} + +#[derive(Debug)] +struct Summary { + reps: usize, + min_us: f64, + p50_us: f64, + p95_us: f64, + mean_us: f64, + max_us: f64, +} + +impl Summary { + fn from_durations(samples: Vec) -> Self { + assert!(!samples.is_empty()); + let mut micros = samples + .into_iter() + .map(|sample| sample.as_secs_f64() * 1_000_000.0) + .collect::>(); + micros.sort_by(f64::total_cmp); + let reps = micros.len(); + let percentile = |p: f64| { + let index = ((reps - 1) as f64 * p).round() as usize; + micros[index] + }; + Self { + reps, + min_us: micros[0], + p50_us: percentile(0.50), + p95_us: percentile(0.95), + mean_us: micros.iter().sum::() / reps as f64, + max_us: micros[reps - 1], + } + } +} + +fn print_summary(suite: &str, case: &str, scale: usize, summary: &Summary, notes: &str) { + println!( + "RESULT\t{suite}\t{case}\t{scale}\t{}\t{:.3}\t{:.3}\t{:.3}\t{:.3}\t{:.3}\t{notes}", + summary.reps, + summary.min_us, + summary.p50_us, + summary.p95_us, + summary.mean_us, + summary.max_us, + ); +} + +async fn measure_async(warmup: usize, reps: usize, mut operation: F) -> Summary +where + F: FnMut() -> Fut, + Fut: Future, +{ + for _ in 0..warmup { + operation().await; + } + let mut samples = Vec::with_capacity(reps); + for _ in 0..reps { + let start = Instant::now(); + operation().await; + samples.push(start.elapsed()); + } + Summary::from_durations(samples) +} + +fn writer_settings() -> Settings { + Settings { + flush_interval: None, + compactor_options: None, + garbage_collector_options: None, + l0_sst_size_bytes: 256 * 1024 * 1024, + l0_max_ssts: 16_384, + l0_max_ssts_per_key: 16_384, + ..Settings::default() + } +} + +fn quiet_reader_options(max_memtable_bytes: u64) -> DbReaderOptions { + DbReaderOptions { + manifest_poll_interval: Duration::from_secs(60 * 60), + checkpoint_lifetime: Duration::from_secs(3 * 60 * 60), + max_memtable_bytes, + ..DbReaderOptions::default() + } +} + +fn polling_reader_options( + max_memtable_bytes: u64, + checkpoint_lifetime: Duration, +) -> DbReaderOptions { + DbReaderOptions { + manifest_poll_interval: POLL_INTERVAL, + checkpoint_lifetime, + max_memtable_bytes, + ..DbReaderOptions::default() + } +} + +fn key_for(index: usize, segmentation: Segmentation) -> Bytes { + match segmentation { + Segmentation::None => Bytes::from(format!("key-{index:08}")), + Segmentation::Fixed4 => Bytes::from(format!("{index:04}-key-{index:08}")), + } +} + +async fn write_wal(db: &Db, index: usize, segmentation: Segmentation) -> Bytes { + let key = key_for(index, segmentation); + db.put_with_options( + &key, + b"value-value-value-value-value", + &PutOptions::default(), + &WriteOptions { + await_durable: false, + ..WriteOptions::default() + }, + ) + .await + .expect("put failed"); + db.flush_with_options(FlushOptions { + flush_type: FlushType::Wal, + }) + .await + .expect("WAL flush failed"); + key +} + +async fn open_db( + path: &str, + store: Arc, + clock: Option>, + segmentation: Segmentation, +) -> Db { + let mut builder = Db::builder(path, store).with_settings(writer_settings()); + if let Some(clock) = clock { + builder = builder.with_system_clock(clock); + } + if matches!(segmentation, Segmentation::Fixed4) { + builder = builder.with_segment_extractor(Arc::new(Fixed4Extractor)); + } + builder.build().await.expect("DB open failed") +} + +async fn open_reader( + path: &str, + store: Arc, + clock: Option>, + segmentation: Segmentation, + options: DbReaderOptions, + recorder: Arc, +) -> DbReader { + let mut builder = DbReader::builder(path, store) + .with_options(options) + .with_metrics_recorder(recorder.clone()) + .with_db_cache_disabled(); + if let Some(clock) = clock { + builder = builder.with_system_clock(clock); + } + if matches!(segmentation, Segmentation::Fixed4) { + builder = builder.with_segment_extractor(Arc::new(Fixed4Extractor)); + } + let reader = builder.build().await.expect("reader open failed"); + // The dispatcher's first ticker fires immediately. Wait for the complete + // startup poll, not merely its first object-store request, so the first + // timed sample cannot race startup work. + wait_for_counter(|| scalar(&recorder, MANIFEST_POLLS), 1).await; + reader +} + +struct ReaderFixture { + path: String, + store: Arc, + db: Db, + reader: Arc, + recorder: Arc, + clock: Option>, + segmentation: Segmentation, + next_key: usize, + latest_key: Option, +} + +impl ReaderFixture { + async fn wal_history( + wal_count: usize, + segmentation: Segmentation, + polling: bool, + checkpoint_lifetime: Duration, + ) -> Self { + let path = format!("bench/db-reader-scaling/{}", Uuid::new_v4()); + let store = Arc::new(InMemory::new()); + let clock = polling.then(|| Arc::new(MockSystemClock::new())); + let db = open_db(&path, Arc::clone(&store), clock.clone(), segmentation).await; + let mut latest_key = None; + for index in 0..wal_count { + latest_key = Some(write_wal(&db, index, segmentation).await); + } + let recorder = Arc::new(DefaultMetricsRecorder::new()); + let options = if polling { + polling_reader_options(1, checkpoint_lifetime) + } else { + quiet_reader_options(1) + }; + let reader = Arc::new( + open_reader( + &path, + Arc::clone(&store), + clock.clone(), + segmentation, + options, + Arc::clone(&recorder), + ) + .await, + ); + Self { + path, + store, + db, + reader, + recorder, + clock, + segmentation, + next_key: wal_count, + latest_key, + } + } + + async fn append_wals(&mut self, count: usize) { + for _ in 0..count { + self.latest_key = Some(write_wal(&self.db, self.next_key, self.segmentation).await); + self.next_key += 1; + } + } + + async fn close(self) { + self.reader.close().await.expect("reader close failed"); + self.db.close().await.expect("DB close failed"); + } +} + +fn scalar(recorder: &DefaultMetricsRecorder, name: &str) -> u64 { + lookup_metric(recorder, name).unwrap_or(0) as u64 +} + +fn object_store_requests(recorder: &DefaultMetricsRecorder, api: &'static str) -> u64 { + lookup_metric_with_labels( + recorder, + instrumented_object_store_stats::REQUEST_COUNT, + &[ + ("component", "reader"), + ("store_type", "main"), + ("op", if api == "put" { "put" } else { "get" }), + ("api", api), + ], + ) + .unwrap_or(0) as u64 +} + +#[derive(Clone, Copy, Default)] +struct ReaderCounters { + lists: u64, + heads: u64, + gets: u64, + get_ranges: u64, + puts: u64, + replay_ssts: u64, +} + +impl ReaderCounters { + fn capture(recorder: &DefaultMetricsRecorder) -> Self { + Self { + lists: object_store_requests(recorder, "list"), + heads: object_store_requests(recorder, "head"), + gets: object_store_requests(recorder, "get"), + get_ranges: object_store_requests(recorder, "get_range"), + puts: object_store_requests(recorder, "put"), + replay_ssts: scalar(recorder, REPLAY_SSTS), + } + } + + fn delta(self, before: Self) -> Self { + Self { + lists: self.lists - before.lists, + heads: self.heads - before.heads, + gets: self.gets - before.gets, + get_ranges: self.get_ranges - before.get_ranges, + puts: self.puts - before.puts, + replay_ssts: self.replay_ssts - before.replay_ssts, + } + } + + fn note(self, reps: usize) -> String { + let reps = reps as f64; + format!( + "list/op={:.2};head/op={:.2};get/op={:.2};range/op={:.2};put/op={:.2};replay_sst/op={:.2}", + self.lists as f64 / reps, + self.heads as f64 / reps, + self.gets as f64 / reps, + self.get_ranges as f64 / reps, + self.puts as f64 / reps, + self.replay_ssts as f64 / reps, + ) + } +} + +async fn wait_for_counter(mut current: F, target: u64) +where + F: FnMut() -> u64, +{ + for _ in 0..100_000 { + if current() >= target { + return; + } + tokio::task::yield_now().await; + } + panic!( + "timed out waiting for reader poll completion: current={}, target={target}", + current() + ); +} + +async fn trigger_poll(fixture: &ReaderFixture, expected_new_wals: u64) { + advance_and_wait_poll(fixture, POLL_INTERVAL, expected_new_wals).await; +} + +async fn advance_and_wait_poll(fixture: &ReaderFixture, advance: Duration, expected_new_wals: u64) { + let clock = fixture + .clock + .as_ref() + .expect("polling fixture has no clock"); + let completed_poll_target = scalar(&fixture.recorder, MANIFEST_POLLS) + 1; + let replay_target = scalar(&fixture.recorder, REPLAY_SSTS) + expected_new_wals; + clock.advance(advance).await; + wait_for_counter( + || scalar(&fixture.recorder, MANIFEST_POLLS), + completed_poll_target, + ) + .await; + if expected_new_wals > 0 { + wait_for_counter(|| scalar(&fixture.recorder, REPLAY_SSTS), replay_target).await; + } +} + +async fn benchmark_usual_paths(scales: &[usize], full: bool) { + println!("SECTION\tusual_paths"); + for &wal_count in scales { + let fixture = ReaderFixture::wal_history( + wal_count, + Segmentation::None, + false, + Duration::from_secs(60), + ) + .await; + let snapshot = fixture.reader.snapshot().await.expect("snapshot failed"); + let latest = fixture + .latest_key + .clone() + .unwrap_or_else(|| Bytes::from_static(b"missing")); + let snapshot_reps = if full { 20_000 } else { 5_000 }; + let read_reps = if full { 2_000 } else { 500 }; + let scan_reps = if full { 250 } else { 75 }; + + let summary = measure_async(200, snapshot_reps, || async { + let snapshot = fixture.reader.snapshot().await.expect("snapshot failed"); + black_box(snapshot.seq()); + }) + .await; + print_summary("usual", "snapshot_create_drop", wal_count, &summary, ""); + + let summary = measure_async(30, read_reps, || async { + black_box(fixture.reader.get(&latest).await.expect("get failed")); + }) + .await; + print_summary("usual", "reader_get_latest", wal_count, &summary, ""); + + let summary = measure_async(30, read_reps, || async { + black_box( + fixture + .reader + .get(b"definitely-missing") + .await + .expect("missing get failed"), + ); + }) + .await; + print_summary("usual", "reader_get_missing", wal_count, &summary, ""); + + let summary = measure_async(30, read_reps, || async { + black_box(snapshot.get(&latest).await.expect("snapshot get failed")); + }) + .await; + print_summary("usual", "snapshot_get_latest", wal_count, &summary, ""); + + let summary = measure_async(10, scan_reps, || async { + let mut iter = fixture.reader.scan(..).await.expect("scan failed"); + black_box(iter.next().await.expect("scan next failed")); + }) + .await; + print_summary("usual", "reader_scan_first", wal_count, &summary, ""); + + let summary = measure_async(10, scan_reps, || async { + let mut iter = snapshot.scan(..).await.expect("snapshot scan failed"); + black_box(iter.next().await.expect("snapshot scan next failed")); + }) + .await; + print_summary("usual", "snapshot_scan_first", wal_count, &summary, ""); + + let full_scan_reps = if wal_count <= 32 { 50 } else { 15 }; + let summary = measure_async(3, full_scan_reps, || async { + let mut iter = fixture.reader.scan(..).await.expect("scan failed"); + let mut count = 0usize; + while iter.next().await.expect("scan next failed").is_some() { + count += 1; + } + black_box(count); + }) + .await; + print_summary("usual", "reader_scan_full", wal_count, &summary, ""); + + drop(snapshot); + fixture.close().await; + } +} + +async fn benchmark_bulk_iterator(full: bool) { + println!("SECTION\tbulk_iterator"); + let row_count = if full { 20_000 } else { 5_000 }; + let path = format!("bench/db-reader-bulk/{}", Uuid::new_v4()); + let store = Arc::new(InMemory::new()); + let db = open_db(&path, Arc::clone(&store), None, Segmentation::None).await; + for index in 0..row_count { + let key = key_for(index, Segmentation::None); + db.put_with_options( + &key, + b"value", + &PutOptions::default(), + &WriteOptions { + await_durable: false, + ..WriteOptions::default() + }, + ) + .await + .expect("put failed"); + } + db.flush_with_options(FlushOptions { + flush_type: FlushType::Wal, + }) + .await + .expect("WAL flush failed"); + let recorder = Arc::new(DefaultMetricsRecorder::new()); + let reader = open_reader( + &path, + Arc::clone(&store), + None, + Segmentation::None, + quiet_reader_options(u64::MAX), + recorder, + ) + .await; + + let summary = measure_async(2, if full { 12 } else { 5 }, || async { + let mut iter = db.scan(..).await.expect("DB scan failed"); + let mut count = 0usize; + while iter.next().await.expect("DB scan next failed").is_some() { + count += 1; + } + assert_eq!(count, row_count); + }) + .await; + print_summary("bulk", "db_scan_one_memtable", row_count, &summary, ""); + + let summary = measure_async(2, if full { 12 } else { 5 }, || async { + let mut iter = reader.scan(..).await.expect("reader scan failed"); + let mut count = 0usize; + while iter + .next() + .await + .expect("reader scan next failed") + .is_some() + { + count += 1; + } + assert_eq!(count, row_count); + }) + .await; + print_summary( + "bulk", + "reader_scan_one_replay_memtable", + row_count, + &summary, + "", + ); + + let snapshot = reader.snapshot().await.expect("snapshot failed"); + let summary = measure_async(2, if full { 12 } else { 5 }, || async { + let mut iter = snapshot.scan(..).await.expect("snapshot scan failed"); + let mut count = 0usize; + while iter + .next() + .await + .expect("snapshot scan next failed") + .is_some() + { + count += 1; + } + assert_eq!(count, row_count); + }) + .await; + print_summary( + "bulk", + "snapshot_scan_one_replay_memtable", + row_count, + &summary, + "", + ); + drop(snapshot); + reader.close().await.expect("reader close failed"); + db.close().await.expect("DB close failed"); +} + +async fn benchmark_iterator_contention(full: bool) { + println!("SECTION\titerator_contention"); + let row_count = if full { 5_000 } else { 2_000 }; + let path = format!("bench/db-reader-iterator-contention/{}", Uuid::new_v4()); + let store = Arc::new(InMemory::new()); + let db = Arc::new(open_db(&path, Arc::clone(&store), None, Segmentation::None).await); + for index in 0..row_count { + db.put_with_options( + &key_for(index, Segmentation::None), + b"value", + &PutOptions::default(), + &WriteOptions { + await_durable: false, + ..WriteOptions::default() + }, + ) + .await + .expect("put failed"); + } + db.flush_with_options(FlushOptions { + flush_type: FlushType::Wal, + }) + .await + .expect("WAL flush failed"); + let reader = Arc::new( + open_reader( + &path, + Arc::clone(&store), + None, + Segmentation::None, + quiet_reader_options(u64::MAX), + Arc::new(DefaultMetricsRecorder::new()), + ) + .await, + ); + + // Populate both read paths' caches before comparing their concurrent scans. + let mut db_warmup = db.scan(..).await.expect("DB warmup scan failed"); + while db_warmup + .next() + .await + .expect("DB warmup next failed") + .is_some() + {} + let mut reader_warmup = reader.scan(..).await.expect("reader warmup scan failed"); + while reader_warmup + .next() + .await + .expect("reader warmup next failed") + .is_some() + {} + + let task_counts = if full { + vec![1, 2, 4, 8, 16, 32] + } else { + vec![1, 4, 16] + }; + for task_count in task_counts { + let barrier = Arc::new(Barrier::new(task_count + 1)); + let mut tasks = Vec::with_capacity(task_count); + for _ in 0..task_count { + let db = Arc::clone(&db); + let barrier = Arc::clone(&barrier); + tasks.push(tokio::spawn(async move { + barrier.wait().await; + let mut iter = db.scan(..).await.expect("DB scan failed"); + let mut count = 0usize; + while iter.next().await.expect("DB scan next failed").is_some() { + count += 1; + } + count + })); + } + barrier.wait().await; + let start = Instant::now(); + for task in tasks { + assert_eq!(task.await.expect("DB scan task failed"), row_count); + } + let elapsed = start.elapsed(); + let total_rows = task_count * row_count; + let summary = Summary::from_durations(vec![elapsed.div_f64(total_rows as f64)]); + print_summary( + "contention", + "db_scan_per_row", + task_count, + &summary, + &format!( + "throughput_rows_s={:.0};total_rows={total_rows}", + total_rows as f64 / elapsed.as_secs_f64() + ), + ); + + let barrier = Arc::new(Barrier::new(task_count + 1)); + let mut tasks = Vec::with_capacity(task_count); + for _ in 0..task_count { + let reader = Arc::clone(&reader); + let barrier = Arc::clone(&barrier); + tasks.push(tokio::spawn(async move { + barrier.wait().await; + let mut iter = reader.scan(..).await.expect("reader scan failed"); + let mut count = 0usize; + while iter + .next() + .await + .expect("reader scan next failed") + .is_some() + { + count += 1; + } + count + })); + } + barrier.wait().await; + let start = Instant::now(); + for task in tasks { + assert_eq!(task.await.expect("reader scan task failed"), row_count); + } + let elapsed = start.elapsed(); + let total_rows = task_count * row_count; + let summary = Summary::from_durations(vec![elapsed.div_f64(total_rows as f64)]); + print_summary( + "contention", + "reader_scan_per_row", + task_count, + &summary, + &format!( + "throughput_rows_s={:.0};total_rows={total_rows}", + total_rows as f64 / elapsed.as_secs_f64() + ), + ); + } + + reader.close().await.expect("reader close failed"); + db.close().await.expect("DB close failed"); +} + +async fn benchmark_incremental_replay(histories: &[usize], deltas: &[usize], full: bool) { + println!("SECTION\tincremental_replay"); + for segmentation in [Segmentation::None, Segmentation::Fixed4] { + for &history in histories { + for &delta in deltas { + let reps = if full { 12 } else { 5 }; + let mut samples = Vec::with_capacity(reps); + let mut counter_total = ReaderCounters::default(); + for _ in 0..reps { + // Rebuild outside the timed interval so every sample starts + // from exactly `history`, rather than accumulating each + // preceding sample's delta into its advertised baseline. + let mut fixture = ReaderFixture::wal_history( + history, + segmentation, + true, + Duration::from_secs(60), + ) + .await; + fixture.append_wals(delta).await; + let before = ReaderCounters::capture(&fixture.recorder); + let start = Instant::now(); + trigger_poll(&fixture, delta as u64).await; + samples.push(start.elapsed()); + let counters = ReaderCounters::capture(&fixture.recorder).delta(before); + counter_total.lists += counters.lists; + counter_total.heads += counters.heads; + counter_total.gets += counters.gets; + counter_total.get_ranges += counters.get_ranges; + counter_total.puts += counters.puts; + counter_total.replay_ssts += counters.replay_ssts; + fixture.close().await; + } + let summary = Summary::from_durations(samples); + let notes = format!("mode={};{}", segmentation.label(), counter_total.note(reps)); + print_summary( + "replay", + "poll_delta", + history * 10_000 + delta, + &summary, + ¬es, + ); + } + } + } +} + +async fn benchmark_generation_rollover(scales: &[usize], full: bool) { + println!("SECTION\tgeneration_rollover"); + for &history in scales.iter().filter(|&&value| value > 0) { + let reps = if full && history <= 32 { 5 } else { 2 }; + let mut samples = Vec::with_capacity(reps); + let mut counter_total = ReaderCounters { + lists: 0, + heads: 0, + gets: 0, + get_ranges: 0, + puts: 0, + replay_ssts: 0, + }; + for _ in 0..reps { + let fixture = ReaderFixture::wal_history( + history, + Segmentation::None, + true, + Duration::from_secs(60), + ) + .await; + let before_manifest = fixture.reader.manifest().id(); + fixture + .db + .flush_with_options(FlushOptions { + flush_type: FlushType::MemTable, + }) + .await + .expect("L0 flush failed"); + let before = ReaderCounters::capture(&fixture.recorder); + let start = Instant::now(); + advance_and_wait_poll(&fixture, POLL_INTERVAL, 0).await; + assert!(fixture.reader.manifest().id() > before_manifest); + samples.push(start.elapsed()); + let delta = ReaderCounters::capture(&fixture.recorder).delta(before); + counter_total.lists += delta.lists; + counter_total.heads += delta.heads; + counter_total.gets += delta.gets; + counter_total.get_ranges += delta.get_ranges; + counter_total.puts += delta.puts; + counter_total.replay_ssts += delta.replay_ssts; + fixture.close().await; + } + let summary = Summary::from_durations(samples); + print_summary( + "rollover", + "flush_all_replayed_wals_to_l0", + history, + &summary, + &counter_total.note(reps), + ); + } +} + +async fn benchmark_final_old_snapshot_drop(scales: &[usize], full: bool) { + println!("SECTION\tfinal_old_snapshot_drop"); + for &history in scales.iter().filter(|&&value| value > 0) { + let reps = if full && history <= 32 { 8 } else { 3 }; + let mut samples = Vec::with_capacity(reps); + for _ in 0..reps { + let fixture = ReaderFixture::wal_history( + history, + Segmentation::None, + true, + Duration::from_secs(60), + ) + .await; + let old_snapshot = fixture.reader.snapshot().await.expect("snapshot failed"); + let before_manifest = fixture.reader.manifest().id(); + fixture + .db + .flush_with_options(FlushOptions { + flush_type: FlushType::MemTable, + }) + .await + .expect("L0 flush failed"); + advance_and_wait_poll(&fixture, POLL_INTERVAL, 0).await; + assert!(fixture.reader.manifest().id() > before_manifest); + + // The reader has moved to an L0-backed generation, so this is the + // last strong owner of the old persistent replay-memtable chain. + let start = Instant::now(); + drop(old_snapshot); + samples.push(start.elapsed()); + fixture.close().await; + } + let summary = Summary::from_durations(samples); + print_summary( + "cleanup", + "drop_final_old_snapshot", + history, + &summary, + "old_generation_replay_memtables", + ); + } +} + +async fn benchmark_snapshot_contention(full: bool) { + println!("SECTION\tsnapshot_contention"); + let fixture = + ReaderFixture::wal_history(128, Segmentation::None, false, Duration::from_secs(60)).await; + let per_task = if full { 25_000 } else { 5_000 }; + let task_counts = if full { + vec![1, 2, 4, 8, 16, 32] + } else { + vec![1, 4, 16] + }; + for task_count in task_counts { + let barrier = Arc::new(Barrier::new(task_count + 1)); + let mut tasks = Vec::with_capacity(task_count); + for _ in 0..task_count { + let reader = Arc::clone(&fixture.reader); + let barrier = Arc::clone(&barrier); + tasks.push(tokio::spawn(async move { + barrier.wait().await; + for _ in 0..per_task { + let snapshot = reader.snapshot().await.expect("snapshot failed"); + black_box(snapshot.seq()); + } + })); + } + barrier.wait().await; + let start = Instant::now(); + for task in tasks { + task.await.expect("snapshot task failed"); + } + let elapsed = start.elapsed(); + let operations = task_count * per_task; + let per_operation = elapsed.div_f64(operations as f64); + let summary = Summary::from_durations(vec![per_operation]); + let throughput = operations as f64 / elapsed.as_secs_f64(); + print_summary( + "contention", + "snapshot_create_drop", + task_count, + &summary, + &format!("throughput_ops_s={throughput:.0};total_ops={operations}"), + ); + } + fixture.close().await; +} + +async fn benchmark_manifest_file_listing(scales: &[usize], full: bool) { + println!("SECTION\tmanifest_file_listing"); + for &file_count in scales { + let fixture = + ReaderFixture::wal_history(0, Segmentation::None, true, Duration::from_secs(60)).await; + let source_id = fixture.reader.manifest().id(); + // Opening a managed reader creates a checkpoint and therefore advances + // the manifest ID. Small requested scales can be below that fixture + // baseline, so report and populate the actual retained-file count. + let target_file_count = file_count.max(source_id as usize); + let source = Path::from(format!( + "{}/manifest/{source_id:020}.manifest", + fixture.path + )); + for id in (source_id + 1)..=target_file_count as u64 { + let destination = Path::from(format!("{}/manifest/{id:020}.manifest", fixture.path)); + fixture + .store + .copy(&source, &destination) + .await + .expect("manifest copy failed"); + } + let reps = if full { 25 } else { 8 }; + for _ in 0..2 { + trigger_poll(&fixture, 0).await; + } + let before = ReaderCounters::capture(&fixture.recorder); + let summary = measure_async(0, reps, || async { + trigger_poll(&fixture, 0).await; + }) + .await; + let counters = ReaderCounters::capture(&fixture.recorder).delta(before); + print_summary( + "manifest", + "no_change_poll_by_retained_files", + target_file_count, + &summary, + &counters.note(reps), + ); + fixture.close().await; + } +} + +async fn benchmark_checkpoint_heavy_manifest(scales: &[usize], full: bool) { + println!("SECTION\tcheckpoint_heavy_manifest"); + for &checkpoint_count in scales { + let path = format!("bench/db-reader-checkpoints/{}", Uuid::new_v4()); + let store = Arc::new(InMemory::new()); + let clock = Arc::new(MockSystemClock::new()); + let db = open_db( + &path, + Arc::clone(&store), + Some(Arc::clone(&clock)), + Segmentation::None, + ) + .await; + for _ in 0..checkpoint_count { + db.create_checkpoint(CheckpointScope::Durable, &CheckpointOptions::default()) + .await + .expect("user checkpoint creation failed"); + } + let recorder = Arc::new(DefaultMetricsRecorder::new()); + let reader = open_reader( + &path, + Arc::clone(&store), + Some(Arc::clone(&clock)), + Segmentation::None, + polling_reader_options(1, Duration::from_secs(1)), + Arc::clone(&recorder), + ) + .await; + let fixture = ReaderFixture { + path, + store, + db, + reader: Arc::new(reader), + recorder, + clock: Some(clock), + segmentation: Segmentation::None, + next_key: 0, + latest_key: None, + }; + + let reps = if full { 15 } else { 5 }; + for _ in 0..2 { + trigger_poll(&fixture, 0).await; + } + let before = ReaderCounters::capture(&fixture.recorder); + let summary = measure_async(0, reps, || async { + trigger_poll(&fixture, 0).await; + }) + .await; + let counters = ReaderCounters::capture(&fixture.recorder).delta(before); + print_summary( + "manifest", + "no_change_poll_by_checkpoint_count", + checkpoint_count, + &summary, + &counters.note(reps), + ); + + let refresh_reps = if full { 7 } else { 3 }; + let before = ReaderCounters::capture(&fixture.recorder); + let mut samples = Vec::with_capacity(refresh_reps); + for _ in 0..refresh_reps { + let start = Instant::now(); + advance_and_wait_poll(&fixture, Duration::from_millis(501), 0).await; + samples.push(start.elapsed()); + } + let counters = ReaderCounters::capture(&fixture.recorder).delta(before); + let summary = Summary::from_durations(samples); + print_summary( + "manifest", + "refresh_one_managed_checkpoint", + checkpoint_count, + &summary, + &counters.note(refresh_reps), + ); + fixture.close().await; + } +} + +async fn build_pinned_generations( + generation_count: usize, +) -> (ReaderFixture, Vec>) { + let fixture = + ReaderFixture::wal_history(0, Segmentation::None, true, Duration::from_secs(60)).await; + let mut snapshots = vec![fixture.reader.snapshot().await.expect("snapshot failed")]; + for index in 1..generation_count { + let key = key_for(index, Segmentation::None); + fixture + .db + .put_with_options( + &key, + b"value", + &PutOptions::default(), + &WriteOptions { + await_durable: false, + ..WriteOptions::default() + }, + ) + .await + .expect("put failed"); + fixture + .db + .flush_with_options(FlushOptions { + flush_type: FlushType::MemTable, + }) + .await + .expect("L0 flush failed"); + let old_manifest = fixture.reader.manifest().id(); + advance_and_wait_poll(&fixture, POLL_INTERVAL, 0).await; + assert!(fixture.reader.manifest().id() > old_manifest); + snapshots.push(fixture.reader.snapshot().await.expect("snapshot failed")); + } + (fixture, snapshots) +} + +async fn benchmark_pinned_generations(scales: &[usize], full: bool) { + println!("SECTION\tpinned_generations"); + for &generation_count in scales { + let (fixture, snapshots) = build_pinned_generations(generation_count).await; + let poll_reps = if full { 15 } else { 5 }; + for _ in 0..2 { + trigger_poll(&fixture, 0).await; + } + let before = ReaderCounters::capture(&fixture.recorder); + let summary = measure_async(0, poll_reps, || async { + trigger_poll(&fixture, 0).await; + }) + .await; + let counters = ReaderCounters::capture(&fixture.recorder).delta(before); + print_summary( + "generation", + "no_change_poll_while_pinned", + generation_count, + &summary, + &counters.note(poll_reps), + ); + + let refresh_reps = if full { 5 } else { 2 }; + let before = ReaderCounters::capture(&fixture.recorder); + let mut refresh_samples = Vec::with_capacity(refresh_reps); + for _ in 0..refresh_reps { + let start = Instant::now(); + advance_and_wait_poll(&fixture, Duration::from_secs(31), 0).await; + refresh_samples.push(start.elapsed()); + } + let counters = ReaderCounters::capture(&fixture.recorder).delta(before); + let summary = Summary::from_durations(refresh_samples); + print_summary( + "generation", + "refresh_all_pinned", + generation_count, + &summary, + &counters.note(refresh_reps), + ); + + drop(snapshots); + let before = ReaderCounters::capture(&fixture.recorder); + let start = Instant::now(); + trigger_poll(&fixture, 0).await; + let summary = Summary::from_durations(vec![start.elapsed()]); + let counters = ReaderCounters::capture(&fixture.recorder).delta(before); + print_summary( + "generation", + "delete_released", + generation_count, + &summary, + &counters.note(1), + ); + fixture.close().await; + } +} + +async fn benchmark_fixed_reader_open(scales: &[usize], full: bool) { + println!("SECTION\tfixed_reader_open"); + for &wal_count in scales { + let path = format!("bench/db-reader-fixed/{}", Uuid::new_v4()); + let store = Arc::new(InMemory::new()); + let db = open_db(&path, Arc::clone(&store), None, Segmentation::None).await; + for index in 0..wal_count { + write_wal(&db, index, Segmentation::None).await; + } + let checkpoint = db + .create_checkpoint(CheckpointScope::Durable, &CheckpointOptions::default()) + .await + .expect("checkpoint creation failed"); + let reps = if full { 10 } else { 3 }; + let mut samples = Vec::with_capacity(reps); + let mut counter_total = ReaderCounters { + lists: 0, + heads: 0, + gets: 0, + get_ranges: 0, + puts: 0, + replay_ssts: 0, + }; + for _ in 0..reps { + let recorder = Arc::new(DefaultMetricsRecorder::new()); + let before = ReaderCounters::capture(&recorder); + let object_store: Arc = store.clone(); + let start = Instant::now(); + let reader = DbReader::builder(path.as_str(), object_store) + .with_checkpoint_id(checkpoint.id) + .with_options(quiet_reader_options(1)) + .with_metrics_recorder(recorder.clone()) + .with_db_cache_disabled() + .build() + .await + .expect("fixed reader open failed"); + samples.push(start.elapsed()); + let delta = ReaderCounters::capture(&recorder).delta(before); + counter_total.lists += delta.lists; + counter_total.heads += delta.heads; + counter_total.gets += delta.gets; + counter_total.get_ranges += delta.get_ranges; + counter_total.puts += delta.puts; + counter_total.replay_ssts += delta.replay_ssts; + reader.close().await.expect("fixed reader close failed"); + } + let summary = Summary::from_durations(samples); + print_summary( + "open", + "fixed_checkpoint_reader", + wal_count, + &summary, + &counter_total.note(reps), + ); + db.close().await.expect("DB close failed"); + } +} + +async fn run() { + let full = std::env::var_os("SLATEDB_READER_BENCH_FULL").is_some(); + println!( + "CONFIG\tprofile={}\tfull={full}", + if cfg!(debug_assertions) { + "debug" + } else { + "release" + } + ); + println!("HEADER\tsuite\tcase\tscale\treps\tmin_us\tp50_us\tp95_us\tmean_us\tmax_us\tnotes"); + + let usual_scales = if full { + vec![0, 1, 8, 32, 128, 512] + } else { + vec![0, 8, 128] + }; + benchmark_usual_paths(&usual_scales, full).await; + benchmark_bulk_iterator(full).await; + benchmark_iterator_contention(full).await; + + let replay_histories = if full { + vec![0, 32, 128, 512] + } else { + vec![0, 128] + }; + let replay_deltas = if full { vec![1, 8, 32] } else { vec![1, 8] }; + benchmark_incremental_replay(&replay_histories, &replay_deltas, full).await; + benchmark_generation_rollover(&usual_scales, full).await; + benchmark_final_old_snapshot_drop(&usual_scales, full).await; + benchmark_snapshot_contention(full).await; + + let manifest_files = if full { + vec![2, 16, 64, 256, 1_024, 4_096] + } else { + vec![2, 64, 1_024] + }; + benchmark_manifest_file_listing(&manifest_files, full).await; + + let checkpoint_counts = if full { + vec![0, 8, 32, 128] + } else { + vec![0, 32] + }; + benchmark_checkpoint_heavy_manifest(&checkpoint_counts, full).await; + + let generation_counts = if full { + vec![1, 8, 32, 128] + } else { + vec![1, 8] + }; + benchmark_pinned_generations(&generation_counts, full).await; + + let fixed_open_scales = if full { + vec![0, 1, 8, 32, 128] + } else { + vec![0, 8, 32] + }; + benchmark_fixed_reader_open(&fixed_open_scales, full).await; +} + +fn main() { + let runtime = tokio::runtime::Builder::new_multi_thread() + .enable_all() + .build() + .expect("failed to build Tokio runtime"); + runtime.block_on(run()); +} diff --git a/slatedb/src/db_cache/mod.rs b/slatedb/src/db_cache/mod.rs index 78f0cca08e..f8dd48fa89 100644 --- a/slatedb/src/db_cache/mod.rs +++ b/slatedb/src/db_cache/mod.rs @@ -46,7 +46,17 @@ pub const DEFAULT_BLOCK_CACHE_CAPACITY: u64 = 512 * 1024 * 1024; pub const DEFAULT_META_CACHE_CAPACITY: u64 = 128 * 1024 * 1024; /// Atomic counter to generate unique scope IDs for `DbCacheWrapper` instances. -static NEXT_CACHE_SCOPE_ID: AtomicU64 = AtomicU64::new(0); +/// Scope `0` belongs exclusively to cache keys serialized before scoping was +/// introduced, so live wrappers start at `1` and can never alias those entries. +static NEXT_CACHE_SCOPE_ID: AtomicU64 = AtomicU64::new(1); + +fn next_cache_scope_id() -> u64 { + NEXT_CACHE_SCOPE_ID + .fetch_update(Ordering::Relaxed, Ordering::Relaxed, |current| { + current.checked_add(1) + }) + .expect("cache scope IDs exhausted") +} /// A `FnOnce` returning a future that produces a [`CachedEntry`] on cache miss. /// @@ -654,7 +664,7 @@ impl DbCacheWrapper { Self { stats: DbCacheStats::new(recorder), cache, - scope_id: NEXT_CACHE_SCOPE_ID.fetch_add(1, Ordering::Relaxed), + scope_id: next_cache_scope_id(), last_err_log_time: Mutex::new(None), system_clock, } @@ -1054,6 +1064,7 @@ pub(crate) mod test_utils { use crate::db_cache::{CachedEntry, CachedKey, DbCache}; use async_trait::async_trait; use std::collections::HashMap; + use std::sync::atomic::{AtomicU64, Ordering}; use std::sync::{Arc, Mutex}; /// A cache that always returns an error from get operations. @@ -1094,43 +1105,72 @@ pub(crate) mod test_utils { pub(crate) struct TestCache { items: Mutex>, + hits: AtomicU64, + misses: AtomicU64, + inserts: AtomicU64, } impl TestCache { pub(crate) fn new() -> Self { Self { items: Mutex::new(HashMap::new()), + hits: AtomicU64::new(0), + misses: AtomicU64::new(0), + inserts: AtomicU64::new(0), } } pub(crate) fn keys(&self) -> Vec { self.items.lock().unwrap().keys().cloned().collect() } + + fn get(&self, key: &CachedKey) -> Option { + let entry = self.items.lock().unwrap().get(key).cloned(); + if entry.is_some() { + self.hits.fetch_add(1, Ordering::Relaxed); + } else { + self.misses.fetch_add(1, Ordering::Relaxed); + } + entry + } + + pub(crate) fn hits(&self) -> u64 { + self.hits.load(Ordering::Relaxed) + } + + pub(crate) fn misses(&self) -> u64 { + self.misses.load(Ordering::Relaxed) + } + + pub(crate) fn inserts(&self) -> u64 { + self.inserts.load(Ordering::Relaxed) + } + + pub(crate) fn clear(&self) { + self.items.lock().unwrap().clear(); + } } #[async_trait] impl DbCache for TestCache { async fn get_block(&self, key: &CachedKey) -> Result, crate::Error> { - let guard = self.items.lock().unwrap(); - Ok(guard.get(key).cloned()) + Ok(self.get(key)) } async fn get_index(&self, key: &CachedKey) -> Result, crate::Error> { - let guard = self.items.lock().unwrap(); - Ok(guard.get(key).cloned()) + Ok(self.get(key)) } async fn get_filter(&self, key: &CachedKey) -> Result, crate::Error> { - let guard = self.items.lock().unwrap(); - Ok(guard.get(key).cloned()) + Ok(self.get(key)) } async fn get_stats(&self, key: &CachedKey) -> Result, crate::Error> { - let guard = self.items.lock().unwrap(); - Ok(guard.get(key).cloned()) + Ok(self.get(key)) } async fn insert(&self, key: CachedKey, value: CachedEntry) { + self.inserts.fetch_add(1, Ordering::Relaxed); let mut guard = self.items.lock().unwrap(); guard.insert(key, value); } @@ -1509,6 +1549,14 @@ mod tests { let shared_cache: Arc = Arc::new(TestCache::new()); let cache_a = DbCacheWrapper::new(shared_cache.clone(), &recorder_a, system_clock.clone()); let cache_b = DbCacheWrapper::new(shared_cache.clone(), &recorder_b, system_clock); + assert_ne!( + cache_a.scope_id, 0, + "live wrappers must not use legacy scope 0" + ); + assert_ne!( + cache_b.scope_id, 0, + "live wrappers must not use legacy scope 0" + ); assert_ne!(cache_a.scope_id, cache_b.scope_id); let policy = BloomFilterPolicy::new(1); diff --git a/slatedb/src/db_iter.rs b/slatedb/src/db_iter.rs index 7d23afb551..f9d5c6b81b 100644 --- a/slatedb/src/db_iter.rs +++ b/slatedb/src/db_iter.rs @@ -15,6 +15,29 @@ use async_trait::async_trait; use bytes::Bytes; use std::collections::VecDeque; use std::ops::RangeBounds; +use std::sync::Arc; + +pub(crate) trait DbIteratorGuard: Send + Sync { + fn enter(&self) -> Result<(), SlateDBError>; + fn exit(&self); +} + +struct DbIteratorGuardPermit { + guard: Arc, +} + +impl DbIteratorGuardPermit { + fn acquire(guard: Arc) -> Result { + guard.enter()?; + Ok(Self { guard }) + } +} + +impl Drop for DbIteratorGuardPermit { + fn drop(&mut self) { + self.guard.exit(); + } +} /// [`DbIteratorRangeTracker`] records the *requested* scan range of a /// [`DbIterator`] so that the transaction manager can detect read-write @@ -183,6 +206,10 @@ pub struct DbIterator { iter: Box, invalidated_error: Option, last_key: Option, + /// Keeps any reader-generation state needed by the iterator alive. The + /// concrete type is intentionally opaque so the common iterator does not + /// depend on `DbReader` internals. + iteration_guard: Option>, } impl DbIterator { @@ -253,9 +280,25 @@ impl DbIterator { iter, invalidated_error: None, last_key: None, + iteration_guard: None, }) } + pub(crate) fn with_iteration_guard(mut self, guard: Arc) -> Self + where + T: DbIteratorGuard + 'static, + { + self.iteration_guard = Some(guard); + self + } + + fn acquire_iteration_guard(&self) -> Result, SlateDBError> { + self.iteration_guard + .as_ref() + .map(|guard| DbIteratorGuardPermit::acquire(Arc::clone(guard))) + .transpose() + } + /// Get the next key-value pair. /// /// This method filters out tombstones and returns the user-facing [`KeyValue`] struct, @@ -278,6 +321,7 @@ impl DbIterator { } pub(crate) async fn next_entry(&mut self) -> Result, SlateDBError> { + let _permit = self.acquire_iteration_guard()?; if let Some(error) = self.invalidated_error.clone() { Err(error) } else { @@ -329,6 +373,7 @@ impl DbIterator { /// /// Returns [`Error`] if the iterator has been invalidated in order to reclaim resources. pub async fn seek>(&mut self, next_key: K) -> Result<(), crate::Error> { + let _permit = self.acquire_iteration_guard().map_err(crate::Error::from)?; let next_key = next_key.as_ref(); if let Some(error) = self.invalidated_error.clone() { Err(error.into()) diff --git a/slatedb/src/db_reader.rs b/slatedb/src/db_reader.rs index 092556552d..41947e060d 100644 --- a/slatedb/src/db_reader.rs +++ b/slatedb/src/db_reader.rs @@ -8,6 +8,7 @@ use { db_cache::CacheTarget, db_cache_manager, db_common::extract_segment_prefix, + db_iter::DbIteratorGuard, db_state::{collect_touched_segments, SsTableId}, db_stats::DbStats, db_status::{ClosedResultWriter, DbStatus, DbStatusManager}, @@ -29,7 +30,7 @@ use { wal::slatedb::store::WalTableStore, wal::WalReader as WalReaderTrait, wal_replay::{WalReplayIterator, WalReplayOptions}, - Checkpoint, DbCacheManagerOps, DbIterator, DbMetadataOps, DbReadOps, + Checkpoint, DbCacheManagerOps, DbIterator, DbMetadataOps, DbReadOps, DbSnapshot, }, async_trait::async_trait, bytes::Bytes, @@ -39,11 +40,15 @@ use { parking_lot::RwLock, slatedb_common::{clock::SystemClock, DbRand}, std::{ - collections::{BTreeSet, VecDeque}, + collections::{BTreeSet, HashMap}, ops::Sub, - sync::{Arc, LazyLock}, + sync::{ + atomic::{AtomicUsize, Ordering}, + Arc, LazyLock, Weak, + }, }, tokio::runtime::Handle, + tokio::sync::Notify, uuid::Uuid, }; @@ -108,12 +113,17 @@ impl WalReplayEnd { /// Read-only interface for accessing a database from either /// the latest persistent state or from an arbitrary checkpoint. +/// +/// A reader that follows the latest state is read-only with respect to user +/// data, but it appends manifest versions to create, refresh, and release GC +/// checkpoints. Opening from an explicit checkpoint does not start that +/// manifest-writing poller. pub struct DbReader { inner: Arc, task_executor: MessageHandlerExecutor, } -struct DbReaderInner { +pub(crate) struct DbReaderInner { manifest_store: Arc, table_store: Arc, wal_reader: Arc, @@ -123,6 +133,7 @@ struct DbReaderInner { system_clock: Arc, oracle: Arc, reader: Reader, + db_stats: DbStats, status_manager: DbStatusManager, segment_extractor: Option>, rand: Arc, @@ -139,15 +150,210 @@ enum DbReaderMessage { } #[derive(Clone)] -struct ReaderState { - manifest_id: u64, - checkpoint: Option, - manifest: Manifest, - imm_memtable: VecDeque>, +pub(crate) struct ReaderState { + generation: Arc, + imm_memtable: ReplayMemtables, last_wal_id: u64, last_remote_persisted_seq: u64, } +struct ReaderGeneration { + manifest_id: u64, + checkpoint: Option>, + manifest: Manifest, + operation_state: AtomicUsize, + operation_drained: Notify, +} + +const GENERATION_INVALID: usize = 1 << (usize::BITS - 1); +const GENERATION_OPERATION_COUNT: usize = !GENERATION_INVALID; + +struct ReaderGenerationPermit { + generation: Arc, +} + +impl Drop for ReaderGenerationPermit { + fn drop(&mut self) { + self.generation.exit_operation(); + } +} + +impl ReaderGeneration { + fn new( + manifest_id: u64, + checkpoint: Option, + manifest: Manifest, + ) -> Arc { + Arc::new(Self { + manifest_id, + checkpoint: checkpoint.map(RwLock::new), + manifest, + operation_state: AtomicUsize::new(0), + operation_drained: Notify::new(), + }) + } + + fn checkpoint(&self) -> Option { + self.checkpoint.as_ref().map(|checkpoint| checkpoint.read().clone()) + } + + fn managed_checkpoint(&self) -> &RwLock { + self.checkpoint + .as_ref() + .expect("managed reader generation must have a checkpoint") + } + + fn invalidate(&self) { + self.operation_state + .fetch_or(GENERATION_INVALID, Ordering::AcqRel); + } + + fn acquire(self: &Arc) -> Result { + self.enter_operation()?; + Ok(ReaderGenerationPermit { + generation: Arc::clone(self), + }) + } + + fn enter_operation(&self) -> Result<(), SlateDBError> { + let mut state = self.operation_state.load(Ordering::Acquire); + loop { + if state & GENERATION_INVALID != 0 { + let checkpoint_id = self + .checkpoint() + .expect("only checkpoint-backed generations can be invalidated") + .id; + return Err(SlateDBError::CheckpointLeaseLost(checkpoint_id)); + } + assert_ne!( + state & GENERATION_OPERATION_COUNT, + GENERATION_OPERATION_COUNT, + "reader generation in-flight operation count overflow" + ); + match self.operation_state.compare_exchange_weak( + state, + state + 1, + Ordering::AcqRel, + Ordering::Acquire, + ) { + Ok(_) => return Ok(()), + Err(current) => state = current, + } + } + } + + fn exit_operation(&self) { + let previous = self.operation_state.fetch_sub(1, Ordering::AcqRel); + assert!( + previous & GENERATION_OPERATION_COUNT > 0, + "reader generation in-flight operation count underflow" + ); + if previous & GENERATION_OPERATION_COUNT == 1 { + self.operation_drained.notify_waiters(); + } + } + + async fn drain(&self) { + loop { + let notified = self.operation_drained.notified(); + tokio::pin!(notified); + notified.as_mut().enable(); + if self.operation_state.load(Ordering::Acquire) & GENERATION_OPERATION_COUNT == 0 { + return; + } + notified.await; + } + } +} + +impl DbIteratorGuard for ReaderGeneration { + fn enter(&self) -> Result<(), SlateDBError> { + self.enter_operation() + } + + fn exit(&self) { + self.exit_operation(); + } +} + +/// Persistent newest-first collection of WAL-replayed immutable memtables. +/// +/// A new replay poll only prepends newly decoded tables and shares the +/// existing tail with snapshots and in-flight reads. This makes applying a WAL +/// delta proportional to the delta rather than to the complete replay history. +#[derive(Clone, Default)] +struct ReplayMemtables { + head: Option>, + len: usize, +} + +struct ReplayMemtableNode { + table: Arc, + older: Option>, +} + +type ReplayPublisher<'a> = &'a mut (dyn FnMut(&ReplayMemtables, u64, u64) + Send); + +impl ReaderState { + pub(crate) fn applied_seq(&self) -> u64 { + self.last_remote_persisted_seq + } +} + +impl ReplayMemtables { + fn prepend(&mut self, table: Arc) { + self.head = Some(Arc::new(ReplayMemtableNode { + table, + older: self.head.take(), + })); + self.len += 1; + } + + fn front(&self) -> Option<&Arc> { + self.head.as_ref().map(|node| &node.table) + } + + fn iter(&self) -> ReplayMemtablesIter<'_> { + ReplayMemtablesIter { + next: self.head.as_deref(), + } + } + + fn len(&self) -> usize { + self.len + } + + #[cfg(test)] + fn is_empty(&self) -> bool { + self.len == 0 + } +} + +impl FromIterator> for ReplayMemtables { + fn from_iter>>(iter: T) -> Self { + let tables = iter.into_iter().collect::>(); + let mut result = Self::default(); + for table in tables.into_iter().rev() { + result.prepend(table); + } + result + } +} + +struct ReplayMemtablesIter<'a> { + next: Option<&'a ReplayMemtableNode>, +} + +impl Iterator for ReplayMemtablesIter<'_> { + type Item = Arc; + + fn next(&mut self) -> Option { + let node = self.next?; + self.next = node.older.as_deref(); + Some(Arc::clone(&node.table)) + } +} + static EMPTY_TABLE: LazyLock> = LazyLock::new(|| Arc::new(KVTable::new())); impl DbStateReader for ReaderState { @@ -155,18 +361,21 @@ impl DbStateReader for ReaderState { Arc::clone(&EMPTY_TABLE) } - fn imm_memtable(&self) -> &VecDeque> { - &self.imm_memtable + fn imm_memtables(&self) -> Box> + '_> { + Box::new(self.imm_memtable.iter()) } fn core(&self) -> &ManifestCore { - &self.manifest.core + &self.generation.manifest.core } } impl From<&ReaderState> for VersionedManifest { fn from(state: &ReaderState) -> Self { - Self::from_manifest(state.manifest_id, state.manifest.clone()) + Self::from_manifest( + state.generation.manifest_id, + state.generation.manifest.clone(), + ) } } @@ -213,18 +422,20 @@ impl DbReaderInner { ), ) }); - + let db_stats = DbStats::new(&recorder); let initial_state = Arc::new( Self::build_reader_state( checkpoint, manifest_id, initial_manifest, - VecDeque::new(), + ReplayMemtables::default(), WalReplayEnd::for_reader(mode, &options), Arc::clone(&table_store), wal_reader.as_ref(), &options, segment_extractor.as_ref(), + None, + &db_stats, ) .await?, ); @@ -251,12 +462,10 @@ impl DbReaderInner { status_manager.clone(), )); - let db_stats = DbStats::new(&recorder); - let state = RwLock::new(initial_state); let reader = Reader::new( Arc::clone(&table_store), - db_stats, + db_stats.clone(), Arc::clone(&mono_clock), oracle.clone(), merge_operator, @@ -272,6 +481,7 @@ impl DbReaderInner { system_clock, oracle, reader, + db_stats, status_manager, segment_extractor, rand, @@ -327,11 +537,26 @@ impl DbReaderInner { ) -> Result, SlateDBError> { self.check_closed()?; let db_state = Arc::clone(&self.state.read()); + let _permit = db_state.generation.acquire()?; self.reader .get_key_value_with_options(key, options, db_state.as_ref(), None, None) .await } + pub(crate) async fn snapshot_get_key_value_with_options + Send>( + &self, + state: Arc, + max_seq: u64, + key: K, + options: &ReadOptions, + ) -> Result, SlateDBError> { + self.check_closed()?; + let _permit = state.generation.acquire()?; + self.reader + .get_key_value_with_options(key, options, state.as_ref(), None, Some(max_seq)) + .await + } + async fn scan_with_options( &self, range: BytesRange, @@ -340,7 +565,9 @@ impl DbReaderInner { ) -> Result { self.check_closed()?; let db_state = Arc::clone(&self.state.read()); - self.reader + let _permit = db_state.generation.acquire()?; + let iter = self + .reader .scan_with_options( range, options, @@ -351,7 +578,34 @@ impl DbReaderInner { prefix, }, ) - .await + .await?; + Ok(iter.with_iteration_guard(Arc::clone(&db_state.generation))) + } + + pub(crate) async fn snapshot_scan_with_options( + &self, + state: Arc, + max_seq: u64, + range: BytesRange, + options: &ScanOptions, + prefix: Option, + ) -> Result { + self.check_closed()?; + let _permit = state.generation.acquire()?; + let iter = self + .reader + .scan_with_options( + range, + options, + ScanContext { + db_state: state.as_ref(), + write_batch_iter: None, + max_seq: Some(max_seq), + prefix, + }, + ) + .await?; + Ok(iter.with_iteration_guard(Arc::clone(&state.generation))) } fn should_reestablish_checkpoint(&self, latest: &ManifestCore) -> bool { @@ -368,24 +622,17 @@ impl DbReaderInner { || latest.segments != current_state.segments } - async fn replace_checkpoint( + async fn create_checkpoint( &self, stored_manifest: &mut StoredManifest, ) -> Result { - let current_checkpoint_id = self - .state - .read() - .checkpoint - .as_ref() - .expect("managed reader must have a checkpoint") - .id; let options = CheckpointOptions { lifetime: Some(self.options.checkpoint_lifetime), ..CheckpointOptions::default() }; let new_checkpoint_id = self.rand.rng().gen_uuid(); stored_manifest - .replace_checkpoint(current_checkpoint_id, new_checkpoint_id, &options) + .write_checkpoint(new_checkpoint_id, &options) .await } @@ -412,33 +659,42 @@ impl DbReaderInner { return Ok(()); } let current_state = Arc::clone(&self.state.read()); - let mut imm_memtable = current_state.imm_memtable().clone(); - let (last_wal_id, last_committed_seq) = Self::replay_wal_into( + let mut imm_memtable = current_state.imm_memtable.clone(); + let generation = Arc::clone(¤t_state.generation); + let mut publish = + |imm_memtable: &ReplayMemtables, last_wal_id: u64, last_committed_seq: u64| { + self.oracle.advance_durable_seq(last_committed_seq); + self.db_stats + .reader_replay_memtables + .set(imm_memtable.len() as i64); + let mut write_guard = self.state.write(); + *write_guard = Arc::new(ReaderState { + generation: Arc::clone(&generation), + imm_memtable: imm_memtable.clone(), + last_wal_id, + last_remote_persisted_seq: last_committed_seq, + }); + drop(write_guard); + self.status_manager + .report_memtable_segments(collect_touched_segments(self.state.read().as_ref())); + }; + + Self::replay_wal_into( Arc::clone(&self.table_store), self.wal_reader.as_ref(), &self.options, current_state.core(), &mut imm_memtable, + Some(( + current_state.last_wal_id, + current_state.last_remote_persisted_seq, + )), WalReplayEnd::Latest, self.segment_extractor.as_ref(), + Some(&mut publish), + Some(&self.db_stats), ) .await?; - - if last_wal_id > current_state.last_wal_id { - self.oracle.advance_durable_seq(last_committed_seq); - let mut write_guard = self.state.write(); - *write_guard = Arc::new(ReaderState { - manifest_id: current_state.manifest_id, - checkpoint: current_state.checkpoint.clone(), - manifest: current_state.manifest.clone(), - imm_memtable, - last_wal_id, - last_remote_persisted_seq: last_committed_seq, - }); - drop(write_guard); - self.status_manager - .report_memtable_segments(collect_touched_segments(self.state.read().as_ref())); - } Ok(()) } @@ -459,7 +715,13 @@ impl DbReaderInner { manifest: Manifest, ) -> Result { let prior = self.state.read().clone(); - let mut imm_memtable = VecDeque::new(); + let replay_cursor = Some(( + prior.last_wal_id.max(manifest.core.replay_after_wal_id), + prior + .last_remote_persisted_seq + .max(manifest.core.last_l0_seq), + )); + let mut retained_memtables = Vec::new(); for table in prior.imm_memtable.iter() { let table_meta = table.table().metadata(); @@ -468,7 +730,7 @@ impl DbReaderInner { continue; } else if table_meta.first_seq > manifest.core.last_l0_seq { // Keep the entire table since all rows are newer than L0+. - imm_memtable.push_back(Arc::clone(table)); + retained_memtables.push(table); } else { // The table has some rows that are newer than L0+ and some that are older. This // happens when the table spans multiple WAL files. Some of those WAL files can @@ -481,10 +743,11 @@ impl DbReaderInner { )?; // Push to the back because we are iterating prior from newest to oldest, and we // want the imm memtables in checkpoint state to be ordered the same way. - imm_memtable.push_back(Arc::new(filtered_table)); + retained_memtables.push(Arc::new(filtered_table)); } } + let imm_memtable = retained_memtables.into_iter().collect(); Self::build_reader_state( checkpoint, manifest_id, @@ -495,6 +758,8 @@ impl DbReaderInner { self.wal_reader.as_ref(), &self.options, self.segment_extractor.as_ref(), + replay_cursor, + &self.db_stats, ) .await } @@ -503,12 +768,14 @@ impl DbReaderInner { checkpoint: Option, manifest_id: u64, manifest: Manifest, - mut imm_memtable: VecDeque>, + mut imm_memtable: ReplayMemtables, replay_wals: Option, table_store: Arc, wal_reader: &dyn WalReaderTrait, options: &DbReaderOptions, segment_extractor: Option<&Arc>, + replay_cursor: Option<(u64, u64)>, + db_stats: &DbStats, ) -> Result { let (last_wal_id, last_committed_seq) = match replay_wals { Some(replay_end) => { @@ -518,8 +785,11 @@ impl DbReaderInner { options, &manifest.core, &mut imm_memtable, + replay_cursor, replay_end, segment_extractor, + None, + Some(db_stats), ) .await? } @@ -528,10 +798,12 @@ impl DbReaderInner { None => Self::replayed_watermark(&manifest.core, &imm_memtable), }; + db_stats + .reader_replay_memtables + .set(imm_memtable.len() as i64); + Ok(ReaderState { - manifest_id, - checkpoint, - manifest, + generation: ReaderGeneration::new(manifest_id, checkpoint, manifest), imm_memtable, last_wal_id, last_remote_persisted_seq: last_committed_seq, @@ -548,7 +820,7 @@ impl DbReaderInner { latest_manifest: VersionedManifest, ) -> Result<(), SlateDBError> { let manifest_id = latest_manifest.id; - if manifest_id <= self.state.read().manifest_id { + if manifest_id <= self.state.read().generation.manifest_id { return self.maybe_replay_new_wals().await; } @@ -560,15 +832,14 @@ impl DbReaderInner { Ok(()) } + #[cfg(test)] async fn maybe_refresh_checkpoint( &self, stored_manifest: &mut StoredManifest, ) -> Result<(), SlateDBError> { - let checkpoint = self - .state - .read() - .checkpoint - .clone() + let generation = Arc::clone(&self.state.read().generation); + let checkpoint = generation + .checkpoint() .expect("managed reader must have a checkpoint"); let half_lifetime = self .options @@ -591,7 +862,7 @@ impl DbReaderInner { // GC reaped it. Re-establish a fresh checkpoint against the latest // manifest instead of failing the reader permanently. warn!("reader checkpoint missing, re-establishing [checkpoint_id={id}]"); - let checkpoint = self.replace_checkpoint(stored_manifest).await?; + let checkpoint = self.create_checkpoint(stored_manifest).await?; self.reestablish_checkpoint(checkpoint).await?; return Ok(()); } @@ -601,20 +872,11 @@ impl DbReaderInner { // Update our local checkpoint copy so we know the latest expiration time // and can calculate future refresh deadlines correctly. { - let mut write_guard = self.state.write(); - let current_state = write_guard.as_ref(); - // Defensively, only update checkpoint if the id and expiry still match. - if current_state - .checkpoint - .as_ref() - .is_some_and(|current_checkpoint| { - current_checkpoint.id == checkpoint.id - && current_checkpoint.expire_time == checkpoint.expire_time - }) + let mut current_checkpoint = generation.managed_checkpoint().write(); + if current_checkpoint.id == checkpoint.id + && current_checkpoint.expire_time == checkpoint.expire_time { - let mut updated_state = current_state.clone(); - updated_state.checkpoint = Some(refreshed_checkpoint.clone()); - *write_guard = Arc::new(updated_state); + *current_checkpoint = refreshed_checkpoint.clone(); } } @@ -630,9 +892,7 @@ impl DbReaderInner { self: &Arc, task_executor: &MessageHandlerExecutor, ) -> Result<(), SlateDBError> { - let poller = ManifestPoller { - inner: Arc::clone(self), - }; + let poller = ManifestPoller::new(Arc::clone(self)); let (_tx, rx) = async_channel::unbounded(); let result = task_executor.add_handler( DB_READER_TASK_NAME.to_string(), @@ -649,7 +909,7 @@ impl DbReaderInner { /// manifest's own boundary when nothing has been replayed into `tables`. fn replayed_watermark( core: &ManifestCore, - tables: &VecDeque>, + tables: &ReplayMemtables, ) -> (u64, u64) { match tables.front() { Some(latest_replayed_table) => ( @@ -665,12 +925,15 @@ impl DbReaderInner { wal_reader: &dyn WalReaderTrait, reader_options: &DbReaderOptions, core: &ManifestCore, - into_tables: &mut VecDeque>, + into_tables: &mut ReplayMemtables, + replay_cursor: Option<(u64, u64)>, replay_end: WalReplayEnd, segment_extractor: Option<&Arc>, + mut publish: Option>, + db_stats: Option<&DbStats>, ) -> Result<(u64, u64), SlateDBError> { - let (mut replay_after_wal_id, mut last_committed_seq) = - Self::replayed_watermark(core, into_tables); + let (mut replay_after_wal_id, mut last_committed_seq) = replay_cursor + .unwrap_or_else(|| Self::replayed_watermark(core, into_tables)); let wal_id_start = replay_after_wal_id .checked_add(1) .ok_or(SlateDBError::InvalidDBState)?; @@ -713,7 +976,16 @@ impl DbReaderInner { // is tagged with the last fully replayed WAL ID, which may equal the // watermark of the previous table. assert!(replayed_table.last_wal_id >= replay_after_wal_id); + let replayed_ssts = replayed_table.last_wal_id - replay_after_wal_id; replay_after_wal_id = replayed_table.last_wal_id; + if let Some(db_stats) = db_stats { + let metadata = replayed_table.table.metadata(); + db_stats.reader_wal_replay_ssts.increment(replayed_ssts); + db_stats + .reader_wal_replay_bytes + .increment(metadata.entries_size_in_bytes as u64); + db_stats.reader_wal_replay_batches.increment(1); + } if !replayed_table.table.is_empty() && replayed_table.last_seq > last_committed_seq { let first_seq = replayed_table .table @@ -732,7 +1004,10 @@ impl DbReaderInner { } let imm_memtable = ImmutableMemtable::new(replayed_table.table, replayed_table.last_wal_id); - into_tables.push_front(Arc::new(imm_memtable)); + into_tables.prepend(Arc::new(imm_memtable)); + } + if let Some(publish) = publish.as_mut() { + publish(into_tables, replay_after_wal_id, last_committed_seq); } } @@ -787,6 +1062,149 @@ impl DbReaderInner { struct ManifestPoller { inner: Arc, + generations: HashMap>, +} + +impl ManifestPoller { + fn new(inner: Arc) -> Self { + let mut generations = HashMap::new(); + if inner.mode == DbReaderMode::ManagedCheckpoint { + let generation = Arc::clone(&inner.state.read().generation); + let checkpoint_id = generation + .checkpoint() + .expect("managed reader must have a checkpoint") + .id; + generations.insert(checkpoint_id, Arc::downgrade(&generation)); + } + let poller = Self { + inner, + generations, + }; + poller.report_active_checkpoints(); + poller + } + + fn report_active_checkpoints(&self) { + self.inner + .db_stats + .reader_active_checkpoints + .set(self.generations.len() as i64); + } + + fn register_current_generation(&mut self) { + let generation = Arc::clone(&self.inner.state.read().generation); + let checkpoint_id = generation + .checkpoint() + .expect("managed reader must have a checkpoint") + .id; + self.generations + .insert(checkpoint_id, Arc::downgrade(&generation)); + self.report_active_checkpoints(); + } + + async fn delete_released_checkpoints( + &mut self, + manifest: &mut StoredManifest, + ) -> Result<(), SlateDBError> { + let released = self + .generations + .iter() + .filter_map(|(id, generation)| generation.upgrade().is_none().then_some(*id)) + .collect::>(); + if released.is_empty() { + return Ok(()); + } + manifest.delete_checkpoints(&released).await?; + for id in released { + self.generations.remove(&id); + } + self.report_active_checkpoints(); + Ok(()) + } + + async fn refresh_live_checkpoints( + &mut self, + manifest: &mut StoredManifest, + ) -> Result<(), SlateDBError> { + let half_lifetime = self + .inner + .options + .checkpoint_lifetime + .checked_div(2) + .expect("checkpoint lifetime division failed"); + loop { + let live = self + .generations + .iter() + .filter_map(|(id, generation)| { + generation.upgrade().map(|generation| (*id, generation)) + }) + .collect::>(); + let now = self.inner.system_clock.now(); + let refresh_due = live.iter().any(|(_, generation)| { + generation + .checkpoint() + .expect("managed reader generation must have a checkpoint") + .expire_time + .is_some_and(|expiry| now > expiry.sub(half_lifetime)) + }); + if !refresh_due { + return Ok(()); + } + + let ids = live.iter().map(|(id, _)| *id).collect::>(); + match manifest + .refresh_checkpoints(&ids, self.inner.options.checkpoint_lifetime) + .await + { + Ok(refreshed) => { + for checkpoint in refreshed { + if let Some(generation) = + self.generations.get(&checkpoint.id).and_then(Weak::upgrade) + { + *generation.managed_checkpoint().write() = checkpoint; + } + } + return Ok(()); + } + Err(SlateDBError::CheckpointMissing(id)) => { + warn!("reader checkpoint lease lost [checkpoint_id={id}]"); + if let Some(generation) = self.generations.remove(&id).and_then(|g| g.upgrade()) + { + generation.invalidate(); + generation.drain().await; + } + self.report_active_checkpoints(); + + let current_id = self + .inner + .state + .read() + .generation + .checkpoint() + .expect("managed reader must have a checkpoint") + .id; + if current_id == id { + let checkpoint = self.inner.create_checkpoint(manifest).await?; + self.inner.reestablish_checkpoint(checkpoint).await?; + self.register_current_generation(); + } + // Other live generations may be due at the same time. Retry + // the batch immediately after removing the missing lease so + // one lost checkpoint cannot make their refresh a poll late. + } + Err(err) => return Err(err), + } + } + } +} + +impl Drop for ManifestPoller { + fn drop(&mut self) { + // The gauge tracks only GC checkpoints actively managed by this + // poller. Reset it even if startup, cleanup, or the poller task fails. + self.inner.db_stats.reader_active_checkpoints.set(0); + } } #[async_trait] @@ -807,24 +1225,30 @@ impl MessageHandler for ManifestPoller { self.inner.system_clock.clone(), ) .await?; + self.delete_released_checkpoints(&mut manifest).await?; let latest_manifest = manifest.manifest(); if self .inner .should_reestablish_checkpoint(&latest_manifest.core) { - let checkpoint = self.inner.replace_checkpoint(&mut manifest).await?; + let checkpoint = self.inner.create_checkpoint(&mut manifest).await?; self.inner.reestablish_checkpoint(checkpoint).await?; + self.register_current_generation(); } else { self.inner.maybe_replay_new_wals().await?; } - self.inner.maybe_refresh_checkpoint(&mut manifest).await + self.refresh_live_checkpoints(&mut manifest).await?; + self.inner.db_stats.reader_manifest_polls.increment(1); + Ok(()) } DbReaderMode::FollowLatest => { let result = self.inner.refresh_latest_manifest().await; if let Err(error) = result { warn!("failed to refresh reader to latest manifest [error={error:?}]"); + } else { + self.inner.db_stats.reader_manifest_polls.increment(1); } Ok(()) } @@ -846,19 +1270,29 @@ impl MessageHandler for ManifestPoller { self.inner.system_clock.clone(), ) .await?; - let checkpoint_id = self - .inner - .state - .read() - .checkpoint - .as_ref() - .expect("managed reader must have a checkpoint") - .id; - info!( - "deleting reader established checkpoint for shutdown [checkpoint_id={}]", - checkpoint_id - ); - manifest.delete_checkpoint(checkpoint_id).await?; + let checkpoint_ids = self.generations.keys().copied().collect::>(); + if !checkpoint_ids.is_empty() { + let live_generations = self + .generations + .values() + .filter_map(Weak::upgrade) + .collect::>(); + // Invalidate every generation before waiting on any one of them, + // otherwise operations could continue entering a later gate while + // shutdown is draining an earlier one. + for generation in &live_generations { + generation.invalidate(); + } + for generation in live_generations { + generation.drain().await; + } + info!( + "deleting reader established checkpoints for shutdown [checkpoint_ids={:?}]", + checkpoint_ids + ); + manifest.delete_checkpoints(&checkpoint_ids).await?; + } + self.inner.db_stats.reader_active_checkpoints.set(0); Ok(()) } } @@ -893,11 +1327,11 @@ impl DbReader { path: Path, ) -> Result<(), SlateDBError> { let state = Arc::clone(&self.inner.state.read()); - let external_ssts = state.manifest.external_ssts(); + let external_ssts = state.generation.manifest.external_ssts(); let path_resolver = PathResolver::new_with_external_ssts(path, external_ssts); let cache_opts = &self.inner.options.object_store_cache_options; crate::utils::preload_cache_from_manifest( - &state.manifest.core, + &state.generation.manifest.core, cached_obj_store, &path_resolver, cache_opts.preload_disk_cache_on_startup, @@ -907,9 +1341,10 @@ impl DbReader { } /// Creates a database reader that can read the contents of a database (but cannot write any - /// data). [`DbReaderMode`] controls whether the reader manages a checkpoint, remains pinned to - /// a supplied checkpoint, or follows the latest manifest without garbage-collection - /// protection. + /// user data). [`DbReaderMode`] controls whether the reader manages GC checkpoints, remains + /// pinned to a supplied checkpoint, or follows the latest manifest without GC protection. + /// Managed readers retain each generation's checkpoint until all snapshots, iterators, and + /// in-flight reads using that generation are gone. pub async fn open>( path: P, object_store: Arc, @@ -924,6 +1359,44 @@ impl DbReader { .await } + /// Captures the reader's latest fully applied state as a read-only + /// snapshot-isolation transaction. + /// + /// This is an O(1) local operation: WAL discovery and replay remain the + /// responsibility of the long-lived reader poller. Creating a snapshot + /// never performs object-store I/O or forces a flush. Snapshots created + /// from the same manifest generation share one GC checkpoint. + pub async fn snapshot(&self) -> Result, crate::Error> { + if self.inner.mode == DbReaderMode::FollowLatest { + return Err(SlateDBError::DbReaderSnapshotUnsupportedInFollowLatest.into()); + } + loop { + self.inner.check_closed()?; + let state = Arc::clone(&self.inner.state.read()); + match state.generation.acquire() { + Ok(_permit) => { + // Keep the generation gate held until the snapshot has + // captured the state. Lease-loss recovery cannot + // invalidate this generation between validation and + // construction. + return Ok(DbSnapshot::new_reader(Arc::clone(&self.inner), state)); + } + Err(err @ SlateDBError::CheckpointLeaseLost(_)) => { + // Recovery may have replaced the invalid generation while + // we were waiting for its gate. Retry only in that case; + // otherwise report the lease loss instead of returning a + // snapshot whose every read is guaranteed to fail. + let current = Arc::clone(&self.inner.state.read()); + if !Arc::ptr_eq(&state, ¤t) { + continue; + } + return Err(err.into()); + } + Err(err) => return Err(err.into()), + } + } + } + /// Creates a new builder for a database reader at the given path. /// /// # Arguments @@ -1371,6 +1844,11 @@ impl DbReader { /// } /// ``` pub async fn close(&self) -> Result<(), crate::Error> { + // Fixed-checkpoint readers do not have a manifest poller to publish + // their clean shutdown, so close the shared status explicitly for + // both reader modes before shutting down any managed task. + self.inner.status_manager.write_result(Ok(())); + self.task_executor .shutdown_task(DB_READER_TASK_NAME) .await @@ -1481,7 +1959,10 @@ impl DbCacheManagerOps for DbReader { mod tests { use crate::wal::slatedb::reader::SlateDbWalReaderOptions; use { - super::{DbReaderMessage, ManifestPoller, ReaderState, WalReplayEnd}, + super::{ + DbReaderMessage, ManifestPoller, ReaderGeneration, ReaderState, ReplayMemtables, + WalReplayEnd, + }, crate::{ block_cache_policy::BlockCachePolicy, clock::MonotonicClock, @@ -1489,8 +1970,9 @@ mod tests { CheckpointOptions, CheckpointScope, CloseOptions, FlushOptions, FlushType, MergeOptions, PutOptions, Settings, WriteOptions, }, + db_cache::{test_utils::TestCache, DbCache}, db_reader::{DbReader, DbReaderInner, DbReaderMode, DbReaderOptions}, - db_state::SstType, + db_state::{SsTableId, SstType}, db_stats::DbStats, db_status::DbStatusManager, dispatcher::MessageHandler, @@ -1567,6 +2049,28 @@ mod tests { } } + async fn wait_for_reader_generation_change(reader: &DbReader, previous: Uuid) { + tokio::time::timeout(Duration::from_secs(5), async { + loop { + if reader + .inner + .state + .read() + .generation + .checkpoint() + .unwrap() + .id + != previous + { + break; + } + tokio::time::sleep(Duration::from_millis(5)).await; + } + }) + .await + .expect("reader did not install a new manifest generation"); + } + #[tokio::test] async fn should_get_value_from_db() { let object_store: Arc = Arc::new(InMemory::new()); @@ -1992,9 +2496,17 @@ mod tests { assert!(latest_manifest.id > initial_manifest_id); assert!(latest_manifest.manifest.core.checkpoints.is_empty()); - let mut poller = ManifestPoller { - inner: Arc::clone(&reader.inner), + let snapshot_error = match reader.snapshot().await { + Ok(_) => panic!("FollowLatest must not create an unprotected snapshot"), + Err(error) => error, }; + assert_eq!(snapshot_error.kind(), crate::ErrorKind::Invalid); + assert!(snapshot_error + .to_string() + .contains("snapshots are unsupported in FollowLatest mode")); + assert!(recording_store.write_kinds().is_empty()); + + let mut poller = ManifestPoller::new(Arc::clone(&reader.inner)); poller.handle(DbReaderMessage::PollManifest).await.unwrap(); assert!(reader.manifest().id() >= latest_manifest.id); @@ -2066,9 +2578,7 @@ mod tests { saved_manifests.push((location, bytes)); } - let mut poller = ManifestPoller { - inner: Arc::clone(&reader.inner), - }; + let mut poller = ManifestPoller::new(Arc::clone(&reader.inner)); poller.handle(DbReaderMessage::PollManifest).await.unwrap(); assert_eq!(reader.manifest().id(), manifest_id); @@ -2346,7 +2856,13 @@ mod tests { ) .await .unwrap(); - let reader_checkpoint_id = inner.state.read().checkpoint.as_ref().unwrap().id; + let reader_checkpoint_id = inner + .state + .read() + .generation + .checkpoint() + .unwrap() + .id; // Simulate the writer's GC reaping the expired checkpoint. let mut stored_manifest = StoredManifest::load(Arc::clone(&manifest_store), clock.clone()) @@ -2368,7 +2884,13 @@ mod tests { .unwrap(); // The reader should have replaced the reaped checkpoint with a new one. - let new_checkpoint_id = inner.state.read().checkpoint.as_ref().unwrap().id; + let new_checkpoint_id = inner + .state + .read() + .generation + .checkpoint() + .unwrap() + .id; assert_ne!(reader_checkpoint_id, new_checkpoint_id); let latest_manifest = manifest_store.read_latest_manifest().await.unwrap(); let checkpoints = &latest_manifest.manifest.core.checkpoints; @@ -2411,15 +2933,562 @@ mod tests { } #[tokio::test] - async fn replay_wal_into_should_use_latest_existing_table_and_keep_newest_first_order() { + async fn reader_snapshot_should_remain_stable_while_wal_replay_advances() { let object_store: Arc = Arc::new(InMemory::new()); - let path = Path::from("/tmp/test_db_reader_replay_order"); + let path = Path::from("/tmp/test_db_reader_snapshot_wal_isolation"); let test_provider = TestProvider::new(path, Arc::clone(&object_store)); - let table_store = test_provider.table_store(); - let wal_store = test_provider.wal_store(); + let db = test_provider + .new_db(Settings { + flush_interval: None, + compactor_options: None, + garbage_collector_options: None, + ..Settings::default() + }) + .await + .unwrap(); - write_wal_sst( - Arc::clone(&wal_store), + let write_options = WriteOptions { + await_durable: false, + ..WriteOptions::default() + }; + db.put_with_options(b"key", b"v1", &PutOptions::default(), &write_options) + .await + .unwrap(); + db.flush_with_options(FlushOptions { + flush_type: FlushType::Wal, + }) + .await + .unwrap(); + + let reader = test_provider + .new_db_reader( + DbReaderOptions { + manifest_poll_interval: Duration::from_millis(10), + ..DbReaderOptions::default() + }, + None, + None, + ) + .await + .unwrap(); + let snapshot = reader.snapshot().await.unwrap(); + + db.put_with_options(b"key", b"v2", &PutOptions::default(), &write_options) + .await + .unwrap(); + db.flush_with_options(FlushOptions { + flush_type: FlushType::Wal, + }) + .await + .unwrap(); + tokio::time::sleep(Duration::from_millis(20)).await; + + assert_eq!(reader.get(b"key").await.unwrap(), Some(Bytes::from("v2"))); + assert_eq!(snapshot.get(b"key").await.unwrap(), Some(Bytes::from("v1"))); + assert_eq!( + reader.snapshot().await.unwrap().get(b"key").await.unwrap(), + Some(Bytes::from("v2")) + ); + } + + #[tokio::test] + async fn reader_snapshot_isolation_should_survive_db_cache_hits_and_eviction() { + let object_store: Arc = Arc::new(InMemory::new()); + let path = Path::from("/tmp/test_db_reader_snapshot_db_cache_isolation"); + let db = Db::builder(path.clone(), Arc::clone(&object_store)) + .with_settings(Settings { + flush_interval: None, + compactor_options: None, + garbage_collector_options: None, + ..Settings::default() + }) + .build() + .await + .unwrap(); + let write_options = WriteOptions { + await_durable: false, + ..WriteOptions::default() + }; + db.put_with_options(b"key", b"v1", &PutOptions::default(), &write_options) + .await + .unwrap(); + db.flush_with_options(FlushOptions { + flush_type: FlushType::MemTable, + }) + .await + .unwrap(); + + let cache = Arc::new(TestCache::new()); + let reader = DbReader::builder(path, Arc::clone(&object_store)) + .with_options(DbReaderOptions { + manifest_poll_interval: Duration::from_millis(10), + ..DbReaderOptions::default() + }) + .with_db_cache(cache.clone()) + .build() + .await + .unwrap(); + let old_snapshot = reader.snapshot().await.unwrap(); + let old_generation = reader + .inner + .state + .read() + .generation + .checkpoint() + .unwrap() + .id; + + assert_eq!( + old_snapshot.get(b"key").await.unwrap(), + Some(Bytes::from_static(b"v1")) + ); + assert!( + cache.inserts() > 0, + "the cold read should populate the cache" + ); + let hits_after_cold_read = cache.hits(); + assert_eq!( + old_snapshot.get(b"key").await.unwrap(), + Some(Bytes::from_static(b"v1")) + ); + assert!( + cache.hits() > hits_after_cold_read, + "the second old-generation read should hit the cache" + ); + + db.put_with_options(b"key", b"v2", &PutOptions::default(), &write_options) + .await + .unwrap(); + db.flush_with_options(FlushOptions { + flush_type: FlushType::MemTable, + }) + .await + .unwrap(); + wait_for_reader_generation_change(&reader, old_generation).await; + + assert_eq!(reader.get(b"key").await.unwrap(), Some(Bytes::from("v2"))); + let hits_before_old_generation_read = cache.hits(); + assert_eq!( + old_snapshot.get(b"key").await.unwrap(), + Some(Bytes::from_static(b"v1")) + ); + assert!( + cache.hits() > hits_before_old_generation_read, + "warming the new SST must not displace or alias the old SST's cache key" + ); + + cache.clear(); + assert_eq!(cache.entry_count(), 0); + let misses_before_reload = cache.misses(); + let inserts_before_reload = cache.inserts(); + assert_eq!( + old_snapshot.get(b"key").await.unwrap(), + Some(Bytes::from_static(b"v1")) + ); + assert!(cache.misses() > misses_before_reload); + assert!( + cache.inserts() > inserts_before_reload, + "an evicted old-generation block should reload from its checkpoint-pinned SST" + ); + assert_eq!( + reader.snapshot().await.unwrap().get(b"key").await.unwrap(), + Some(Bytes::from_static(b"v2")) + ); + + drop(old_snapshot); + reader.close().await.unwrap(); + db.close().await.unwrap(); + } + + #[tokio::test] + async fn reader_snapshot_isolation_should_survive_warm_object_store_cache_hits() { + use crate::cached_object_store::stats::PART_HIT_COUNT; + use slatedb_common::metrics::{lookup_metric, DefaultMetricsRecorder}; + + let object_store: Arc = Arc::new(InMemory::new()); + let path = Path::from("/tmp/test_db_reader_snapshot_object_cache_isolation"); + let db = Db::builder(path.clone(), Arc::clone(&object_store)) + .with_settings(Settings { + flush_interval: None, + compactor_options: None, + garbage_collector_options: None, + ..Settings::default() + }) + .build() + .await + .unwrap(); + let write_options = WriteOptions { + await_durable: false, + ..WriteOptions::default() + }; + db.put_with_options(b"key", b"v1", &PutOptions::default(), &write_options) + .await + .unwrap(); + db.flush_with_options(FlushOptions { + flush_type: FlushType::MemTable, + }) + .await + .unwrap(); + + let cache_dir = tempfile::Builder::new() + .prefix("dbreader_snapshot_object_cache_") + .tempdir() + .unwrap(); + let mut options = DbReaderOptions { + manifest_poll_interval: Duration::from_millis(10), + ..DbReaderOptions::default() + }; + options.object_store_cache_options.root_folder = Some(cache_dir.path().to_path_buf()); + options.object_store_cache_options.part_size_bytes = 1024; + options.object_store_cache_options.scan_interval = None; + let metrics = Arc::new(DefaultMetricsRecorder::new()); + let reader = DbReader::builder(path, Arc::clone(&object_store)) + .with_options(options) + // Force SST reads through the object-store cache instead of allowing + // the in-memory block cache to satisfy the second read first. + .with_db_cache_disabled() + .with_metrics_recorder(metrics.clone()) + .build() + .await + .unwrap(); + let old_snapshot = reader.snapshot().await.unwrap(); + let old_generation = reader + .inner + .state + .read() + .generation + .checkpoint() + .unwrap() + .id; + + assert_eq!( + old_snapshot.get(b"key").await.unwrap(), + Some(Bytes::from_static(b"v1")) + ); + let hits_after_cold_read = lookup_metric(&metrics, PART_HIT_COUNT).unwrap_or(0); + assert_eq!( + old_snapshot.get(b"key").await.unwrap(), + Some(Bytes::from_static(b"v1")) + ); + assert!( + lookup_metric(&metrics, PART_HIT_COUNT).unwrap_or(0) > hits_after_cold_read, + "the second read should be served from the local object-store cache" + ); + + db.put_with_options(b"key", b"v2", &PutOptions::default(), &write_options) + .await + .unwrap(); + db.flush_with_options(FlushOptions { + flush_type: FlushType::MemTable, + }) + .await + .unwrap(); + wait_for_reader_generation_change(&reader, old_generation).await; + assert_eq!(reader.get(b"key").await.unwrap(), Some(Bytes::from("v2"))); + + let hits_before_old_generation_read = lookup_metric(&metrics, PART_HIT_COUNT).unwrap_or(0); + assert_eq!( + old_snapshot.get(b"key").await.unwrap(), + Some(Bytes::from_static(b"v1")) + ); + assert!( + lookup_metric(&metrics, PART_HIT_COUNT).unwrap_or(0) > hits_before_old_generation_read, + "the old snapshot should still use the cached bytes for its own immutable SST" + ); + + drop(old_snapshot); + reader.close().await.unwrap(); + db.close().await.unwrap(); + } + + #[tokio::test] + async fn reader_snapshot_creation_should_do_no_io_and_share_generation_checkpoint() { + let recording = Arc::new(test_utils::RecordingObjectStore::new(Arc::new( + InMemory::new(), + ))); + let object_store: Arc = recording.clone(); + let path = Path::from("/tmp/test_db_reader_snapshot_no_io"); + let test_provider = TestProvider::new(path, Arc::clone(&object_store)); + let db = test_provider.new_db(Settings::default()).await.unwrap(); + db.put(b"key", b"value").await.unwrap(); + db.flush().await.unwrap(); + let checkpoint = db + .create_checkpoint(CheckpointScope::All, &CheckpointOptions::default()) + .await + .unwrap(); + let reader = test_provider + .new_db_reader(DbReaderOptions::default(), Some(checkpoint.id), None) + .await + .unwrap(); + recording.clear(); + + let snapshots = futures::future::try_join_all((0..100).map(|_| reader.snapshot())) + .await + .unwrap(); + + assert_eq!(100, snapshots.len()); + assert!(recording.get_kinds(false).is_empty()); + assert!(recording.get_kinds(true).is_empty()); + assert!(recording.write_kinds().is_empty()); + let manifest = test_provider + .manifest_store() + .read_latest_manifest() + .await + .unwrap(); + assert_eq!(1, manifest.manifest.core.checkpoints.len()); + } + + #[tokio::test] + async fn reader_snapshot_should_reject_an_invalid_current_generation() { + let object_store: Arc = Arc::new(InMemory::new()); + let path = Path::from("/tmp/test_db_reader_snapshot_invalid_generation"); + let test_provider = TestProvider::new(path, Arc::clone(&object_store)); + let db = test_provider.new_db(Settings::default()).await.unwrap(); + let checkpoint = db + .create_checkpoint(CheckpointScope::All, &CheckpointOptions::default()) + .await + .unwrap(); + let reader = test_provider + .new_db_reader(DbReaderOptions::default(), Some(checkpoint.id), None) + .await + .unwrap(); + reader.inner.state.read().generation.invalidate(); + + let result = reader.snapshot().await; + + let err = match result { + Ok(_) => panic!("an invalid generation must not produce a snapshot"), + Err(err) => err, + }; + assert!(err.to_string().contains("reader checkpoint lease lost")); + } + + #[tokio::test] + async fn snapshot_should_keep_old_generation_checkpoint_until_drop() { + let object_store: Arc = Arc::new(InMemory::new()); + let path = Path::from("/tmp/test_db_reader_snapshot_checkpoint_retention"); + let test_provider = TestProvider::new(path, Arc::clone(&object_store)); + let db = test_provider + .new_db(Settings { + flush_interval: None, + compactor_options: None, + garbage_collector_options: None, + ..Settings::default() + }) + .await + .unwrap(); + let write_options = WriteOptions { + await_durable: false, + ..WriteOptions::default() + }; + db.put_with_options(b"key", b"v1", &PutOptions::default(), &write_options) + .await + .unwrap(); + db.flush_with_options(FlushOptions { + flush_type: FlushType::MemTable, + }) + .await + .unwrap(); + + let reader = test_provider + .new_db_reader( + DbReaderOptions { + manifest_poll_interval: Duration::from_millis(10), + ..DbReaderOptions::default() + }, + None, + None, + ) + .await + .unwrap(); + let snapshot = reader.snapshot().await.unwrap(); + let old_checkpoint_id = reader + .inner + .state + .read() + .generation + .checkpoint() + .unwrap() + .id; + + db.put_with_options(b"key", b"v2", &PutOptions::default(), &write_options) + .await + .unwrap(); + db.flush_with_options(FlushOptions { + flush_type: FlushType::MemTable, + }) + .await + .unwrap(); + tokio::time::sleep(Duration::from_millis(20)).await; + + let new_checkpoint_id = reader + .inner + .state + .read() + .generation + .checkpoint() + .unwrap() + .id; + assert_ne!(old_checkpoint_id, new_checkpoint_id); + let manifest = test_provider + .manifest_store() + .read_latest_manifest() + .await + .unwrap(); + assert!(manifest + .manifest + .core + .find_checkpoint(old_checkpoint_id) + .is_some()); + assert!(manifest + .manifest + .core + .find_checkpoint(new_checkpoint_id) + .is_some()); + assert_eq!(snapshot.get(b"key").await.unwrap(), Some(Bytes::from("v1"))); + + drop(snapshot); + tokio::time::sleep(Duration::from_millis(20)).await; + + let manifest = test_provider + .manifest_store() + .read_latest_manifest() + .await + .unwrap(); + assert!(manifest + .manifest + .core + .find_checkpoint(old_checkpoint_id) + .is_none()); + assert!(manifest + .manifest + .core + .find_checkpoint(new_checkpoint_id) + .is_some()); + } + + #[tokio::test] + async fn snapshot_iterator_should_retain_checkpoint_after_snapshot_drop() { + let object_store: Arc = Arc::new(InMemory::new()); + let path = Path::from("/tmp/test_db_reader_snapshot_iterator_retention"); + let test_provider = TestProvider::new(path, Arc::clone(&object_store)); + let db = test_provider + .new_db(Settings { + flush_interval: None, + compactor_options: None, + garbage_collector_options: None, + ..Settings::default() + }) + .await + .unwrap(); + let write_options = WriteOptions { + await_durable: false, + ..WriteOptions::default() + }; + db.put_with_options(b"a", b"old", &PutOptions::default(), &write_options) + .await + .unwrap(); + db.flush_with_options(FlushOptions { + flush_type: FlushType::MemTable, + }) + .await + .unwrap(); + + let reader = test_provider + .new_db_reader( + DbReaderOptions { + manifest_poll_interval: Duration::from_millis(10), + ..DbReaderOptions::default() + }, + None, + None, + ) + .await + .unwrap(); + let snapshot = reader.snapshot().await.unwrap(); + let mut iter = snapshot.scan(..).await.unwrap(); + let old_checkpoint_id = reader + .inner + .state + .read() + .generation + .checkpoint() + .unwrap() + .id; + drop(snapshot); + + db.put_with_options(b"b", b"new", &PutOptions::default(), &write_options) + .await + .unwrap(); + db.flush_with_options(FlushOptions { + flush_type: FlushType::MemTable, + }) + .await + .unwrap(); + tokio::time::sleep(Duration::from_millis(20)).await; + + let manifest = test_provider + .manifest_store() + .read_latest_manifest() + .await + .unwrap(); + assert!(manifest + .manifest + .core + .find_checkpoint(old_checkpoint_id) + .is_some()); + let row = iter.next().await.unwrap().unwrap(); + assert_eq!(row.key, Bytes::from("a")); + assert_eq!(row.value, Bytes::from("old")); + assert!(iter.next().await.unwrap().is_none()); + + drop(iter); + tokio::time::sleep(Duration::from_millis(20)).await; + let manifest = test_provider + .manifest_store() + .read_latest_manifest() + .await + .unwrap(); + assert!(manifest + .manifest + .core + .find_checkpoint(old_checkpoint_id) + .is_none()); + } + + #[tokio::test] + async fn generation_drain_should_wait_for_in_flight_operations_and_reject_new_ones() { + let clock: Arc = Arc::new(DefaultSystemClock::new()); + let generation = ReaderGeneration::new( + 1, + Some(test_checkpoint(1, clock)), + Manifest::initial(ManifestCore::new()), + ); + let permit = generation.acquire().unwrap(); + generation.invalidate(); + let draining_generation = Arc::clone(&generation); + let drain = tokio::spawn(async move { draining_generation.drain().await }); + tokio::task::yield_now().await; + assert!(!drain.is_finished()); + + drop(permit); + drain.await.unwrap(); + assert!(matches!( + generation.acquire(), + Err(SlateDBError::CheckpointLeaseLost(_)) + )); + } + + #[tokio::test] + async fn replay_wal_into_should_use_latest_existing_table_and_keep_newest_first_order() { + let object_store: Arc = Arc::new(InMemory::new()); + let path = Path::from("/tmp/test_db_reader_replay_order"); + let test_provider = TestProvider::new(path, Arc::clone(&object_store)); + let table_store = test_provider.table_store(); + let wal_store = test_provider.wal_store(); + + write_wal_sst( + Arc::clone(&wal_store), 3, vec![RowEntry::new_value(b"stale_key", b"stale_value", 3)], ) @@ -2433,15 +3502,18 @@ mod tests { .await .unwrap(); - let mut into_tables = VecDeque::new(); - into_tables.push_front(immutable_memtable( - 3, - vec![RowEntry::new_value(b"stale_key", b"stale_value", 3)], - )); - into_tables.push_back(immutable_memtable( - 2, - vec![RowEntry::new_value(b"older_key", b"older_value", 2)], - )); + let mut into_tables: ReplayMemtables = [ + immutable_memtable( + 3, + vec![RowEntry::new_value(b"stale_key", b"stale_value", 3)], + ), + immutable_memtable( + 2, + vec![RowEntry::new_value(b"older_key", b"older_value", 2)], + ), + ] + .into_iter() + .collect(); let mut core = ManifestCore::new(); core.next_wal_sst_id = 5; @@ -2453,8 +3525,11 @@ mod tests { &DbReaderOptions::default(), &core, &mut into_tables, + None, WalReplayEnd::Manifest, None, + None, + None, ) .await .unwrap(); @@ -2474,6 +3549,76 @@ mod tests { .await; } + #[test] + fn replay_memtable_prepend_should_share_the_existing_tail() { + let mut original = ReplayMemtables::default(); + original.prepend(immutable_memtable( + 1, + vec![RowEntry::new_value(b"old", b"value", 1)], + )); + let original_head = Arc::clone(original.head.as_ref().unwrap()); + + let mut extended = original.clone(); + extended.prepend(immutable_memtable( + 2, + vec![RowEntry::new_value(b"new", b"value", 2)], + )); + + let shared_tail = extended.head.as_ref().unwrap().older.as_ref().unwrap(); + assert!(Arc::ptr_eq(shared_tail, &original_head)); + assert_eq!(1, original.len()); + assert_eq!(2, extended.len()); + } + + #[tokio::test] + async fn replay_wal_into_should_publish_each_bounded_batch() { + let object_store: Arc = Arc::new(InMemory::new()); + let path = Path::from("/tmp/test_db_reader_incremental_replay_publication"); + let test_provider = TestProvider::new(path, Arc::clone(&object_store)); + let table_store = test_provider.table_store(); + let wal_store = test_provider.wal_store(); + let rows = (1..=3) + .map(|seq| RowEntry::new_value(format!("key-{seq}").as_bytes(), &[b'x'; 128], seq)) + .collect::>(); + for (wal_id, row) in rows.iter().cloned().enumerate() { + write_wal_sst(Arc::clone(&wal_store), wal_id as u64 + 1, vec![row]) + .await + .unwrap(); + } + let max_memtable_bytes = + table_store.estimate_encoded_size_compacted(1, rows[0].estimated_size()) as u64; + let options = DbReaderOptions { + max_memtable_bytes, + ..DbReaderOptions::default() + }; + let mut core = ManifestCore::new(); + core.next_wal_sst_id = 4; + let status_manager = status_manager_for_core(&core); + let mut into_tables = ReplayMemtables::default(); + let mut publications = Vec::new(); + let mut publish = |tables: &ReplayMemtables, wal_id, seq| { + publications.push((wal_id, seq, tables.len())); + }; + + let result = DbReaderInner::replay_wal_into( + Arc::clone(&table_store), + &native_wal_reader(&wal_store, &status_manager), + &options, + &core, + &mut into_tables, + None, + WalReplayEnd::Manifest, + None, + Some(&mut publish), + None, + ) + .await + .unwrap(); + + assert_eq!((3, 3), result); + assert_eq!(vec![(1, 1, 1), (2, 2, 2), (3, 3, 3)], publications); + } + #[tokio::test] async fn replay_wal_into_should_treat_missing_wal_sst_as_end_of_iteration() { let object_store: Arc = Arc::new(InMemory::new()); @@ -2490,7 +3635,7 @@ mod tests { .await .unwrap(); - let mut into_tables = VecDeque::new(); + let mut into_tables = ReplayMemtables::default(); let mut core = ManifestCore::new(); core.next_wal_sst_id = 3; let status_manager = status_manager_for_core(&core); @@ -2501,8 +3646,11 @@ mod tests { &DbReaderOptions::default(), &core, &mut into_tables, + None, WalReplayEnd::Manifest, None, + None, + None, ) .await .unwrap(); @@ -2543,7 +3691,7 @@ mod tests { .await .unwrap(); - let mut into_tables = VecDeque::new(); + let mut into_tables = ReplayMemtables::default(); let mut core = ManifestCore::new(); // Force the reader to attempt to read up to 4 even though 3 and 4 don't exist. core.next_wal_sst_id = 4; @@ -2559,8 +3707,11 @@ mod tests { &reader_options, &core, &mut into_tables, + None, WalReplayEnd::Manifest, None, + None, + None, ) .await .unwrap(); @@ -2588,7 +3739,7 @@ mod tests { let table_store = test_provider.table_store(); let wal_store = test_provider.wal_store(); - let mut into_tables = VecDeque::new(); + let mut into_tables = ReplayMemtables::default(); let core = ManifestCore::new(); let status_manager = status_manager_for_core(&core); @@ -2598,8 +3749,11 @@ mod tests { &DbReaderOptions::default(), &core, &mut into_tables, + None, WalReplayEnd::Latest, None, + None, + None, ) .await .unwrap(); @@ -2622,7 +3776,7 @@ mod tests { .await .unwrap(); - let mut into_tables = VecDeque::new(); + let mut into_tables = ReplayMemtables::default(); let core = ManifestCore::new(); let status_manager = status_manager_for_core(&core); @@ -2632,8 +3786,11 @@ mod tests { &DbReaderOptions::default(), &core, &mut into_tables, + None, WalReplayEnd::Latest, None, + None, + None, ) .await .unwrap(); @@ -2661,8 +3818,8 @@ mod tests { .await .unwrap(); - let mut into_tables = VecDeque::new(); - into_tables.push_front(immutable_memtable( + let mut into_tables = ReplayMemtables::default(); + into_tables.prepend(immutable_memtable( 5, vec![ RowEntry::new_value(b"existing_key_1", b"existing_value_1", 9), @@ -2681,18 +3838,48 @@ mod tests { &DbReaderOptions::default(), &core, &mut into_tables, + None, + WalReplayEnd::Latest, + None, + None, + None, + ) + .await + .unwrap(); + + assert_eq!(last_wal_id, 6); + assert_eq!(last_committed_seq, 10); + + let head_after_first_replay = Arc::clone(into_tables.head.as_ref().unwrap()); + let (last_wal_id, last_committed_seq) = DbReaderInner::replay_wal_into( + Arc::clone(&table_store), + &native_wal_reader(&wal_store, &status_manager), + &DbReaderOptions::default(), + &core, + &mut into_tables, + Some((last_wal_id, last_committed_seq)), WalReplayEnd::Latest, None, + None, + None, ) .await .unwrap(); assert_eq!(last_wal_id, 6); assert_eq!(last_committed_seq, 10); + assert!(Arc::ptr_eq( + into_tables.head.as_ref().unwrap(), + &head_after_first_replay + )); } #[tokio::test(start_paused = true)] async fn should_fail_new_reads_if_manifest_poller_crashes() { + use slatedb_common::metrics::{ + lookup_metric, DefaultMetricsRecorder, MetricLevel, MetricsRecorderHelper, + }; + let object_store: Arc = Arc::new(InMemory::new()); let path = Path::from("/tmp/test_kv_store"); let test_provider = TestProvider::new(path.clone(), Arc::clone(&object_store)); @@ -2702,10 +3889,27 @@ mod tests { manifest_poll_interval: Duration::from_millis(500), ..DbReaderOptions::default() }; - let reader = test_provider - .new_db_reader(reader_options, None, None) - .await - .unwrap(); + let metrics_recorder = Arc::new(DefaultMetricsRecorder::new()); + let reader = DbReader::open_internal( + test_provider.manifest_store(), + test_provider.table_store(), + DbReaderMode::ManagedCheckpoint, + None, + None, + reader_options, + Arc::clone(&test_provider.system_clock), + Arc::clone(&test_provider.rand), + MetricsRecorderHelper::new(metrics_recorder.clone(), MetricLevel::default()), + ) + .await + .unwrap(); + assert_eq!( + Some(1), + lookup_metric( + &metrics_recorder, + crate::db_stats::READER_ACTIVE_CHECKPOINTS + ) + ); fail_parallel::cfg( Arc::clone(&test_provider.fp_registry), @@ -2715,11 +3919,17 @@ mod tests { .unwrap(); tokio::time::sleep(Duration::from_millis(20)).await; let result = reader.get(b"key").await.unwrap_err(); - dbg!(&result); assert_eq!( result.to_string(), "Unavailable error: wal unavailable (io error)" ); + assert_eq!( + Some(0), + lookup_metric( + &metrics_recorder, + crate::db_stats::READER_ACTIVE_CHECKPOINTS + ) + ); } #[tokio::test] @@ -3325,12 +4535,14 @@ mod tests { // Seed the prior checkpoint state with IMMs. let input_tables: Vec<_> = case.tables.iter().map(InputMemtable::build).collect(); let prior_state = ReaderState { - manifest_id: stored_manifest.id(), - checkpoint: Some(test_checkpoint( + generation: ReaderGeneration::new( stored_manifest.id(), - test_provider.system_clock.clone(), - )), - manifest: stored_manifest.manifest().clone(), + Some(test_checkpoint( + stored_manifest.id(), + test_provider.system_clock.clone(), + )), + stored_manifest.manifest().clone(), + ), imm_memtable: input_tables.iter().cloned().collect(), last_wal_id: 0, last_remote_persisted_seq: 0, @@ -3356,9 +4568,10 @@ mod tests { // directly. skip_wal_replay keeps the test scoped to the IMM retention logic. let oracle = Arc::new(DbReaderOracle::new(0, DbStatusManager::new(0))); let recorder = slatedb_common::metrics::MetricsRecorderHelper::noop(); + let db_stats = DbStats::new(&recorder); let reader = Reader::new( Arc::clone(&table_store), - DbStats::new(&recorder), + db_stats.clone(), Arc::new(MonotonicClock::new( test_provider.system_clock.clone(), i64::MIN, @@ -3381,6 +4594,7 @@ mod tests { system_clock: test_provider.system_clock.clone(), oracle, reader, + db_stats, status_manager, segment_extractor: None, rand: test_provider.rand.clone(), @@ -3396,7 +4610,10 @@ mod tests { .unwrap(); // The rebuilt checkpoint should reflect the new manifest. - assert_eq!(rebuilt_state.manifest.core.last_l0_seq, case.last_l0_seq); + assert_eq!( + rebuilt_state.generation.manifest.core.last_l0_seq, + case.last_l0_seq + ); assert_eq!(rebuilt_state.imm_memtable.len(), case.expected.len()); for (rebuilt_table, expected_table) in @@ -3434,22 +4651,27 @@ mod tests { let status_manager = status_manager_for_core(current_core); let prior_state = ReaderState { - manifest_id: 1, - checkpoint: Some(test_checkpoint(1, test_provider.system_clock.clone())), - manifest: Manifest::initial(current_core.clone()), - imm_memtable: VecDeque::from([immutable_memtable( + generation: ReaderGeneration::new( + 1, + Some(test_checkpoint(1, test_provider.system_clock.clone())), + Manifest::initial(current_core.clone()), + ), + imm_memtable: [immutable_memtable( 1, vec![RowEntry::new_value(b"key", b"value", 10)], - )]), + )] + .into_iter() + .collect(), last_wal_id: 1, last_remote_persisted_seq: 10, }; let oracle = Arc::new(DbReaderOracle::new(0, DbStatusManager::new(0))); let recorder = slatedb_common::metrics::MetricsRecorderHelper::noop(); + let db_stats = DbStats::new(&recorder); let reader = Reader::new( Arc::clone(&table_store), - DbStats::new(&recorder), + db_stats.clone(), Arc::new(MonotonicClock::new( test_provider.system_clock.clone(), i64::MIN, @@ -3468,6 +4690,7 @@ mod tests { system_clock: test_provider.system_clock.clone(), oracle, reader, + db_stats, status_manager, segment_extractor: None, rand: test_provider.rand.clone(), @@ -3653,7 +4876,9 @@ mod tests { #[tokio::test] async fn should_record_metrics_with_recorder() { - use slatedb_common::metrics::{lookup_metric_with_labels, DefaultMetricsRecorder}; + use slatedb_common::metrics::{ + lookup_metric, lookup_metric_with_labels, DefaultMetricsRecorder, + }; let object_store: Arc = Arc::new(InMemory::new()); let path = Path::from("/tmp/test_db_reader_metrics"); @@ -3687,6 +4912,180 @@ mod tests { ), Some(1) ); + for _ in 0..1_000 { + if lookup_metric(&metrics_recorder, crate::db_stats::READER_MANIFEST_POLLS) == Some(1) { + break; + } + tokio::task::yield_now().await; + } + assert_eq!( + lookup_metric(&metrics_recorder, crate::db_stats::READER_MANIFEST_POLLS,), + Some(1) + ); + } + + #[tokio::test] + async fn should_record_incremental_wal_replay_metrics() { + use slatedb_common::metrics::{lookup_metric, DefaultMetricsRecorder}; + + let object_store: Arc = Arc::new(InMemory::new()); + let path = Path::from("/tmp/test_db_reader_wal_replay_metrics"); + let db = Db::builder(path.clone(), Arc::clone(&object_store)) + .with_settings(Settings { + flush_interval: None, + compactor_options: None, + garbage_collector_options: None, + ..Settings::default() + }) + .build() + .await + .unwrap(); + db.put_with_options( + b"key", + b"value", + &PutOptions::default(), + &WriteOptions { + await_durable: false, + ..WriteOptions::default() + }, + ) + .await + .unwrap(); + db.flush_with_options(FlushOptions { + flush_type: FlushType::Wal, + }) + .await + .unwrap(); + + let metrics_recorder = Arc::new(DefaultMetricsRecorder::new()); + let reader = DbReader::builder(path, object_store) + .with_metrics_recorder(metrics_recorder.clone()) + .build() + .await + .unwrap(); + + assert!( + lookup_metric(&metrics_recorder, crate::db_stats::READER_WAL_REPLAY_SSTS) + .is_some_and(|value| value > 0) + ); + assert!( + lookup_metric(&metrics_recorder, crate::db_stats::READER_WAL_REPLAY_BYTES) + .is_some_and(|value| value > 0) + ); + assert!(lookup_metric( + &metrics_recorder, + crate::db_stats::READER_WAL_REPLAY_BATCHES + ) + .is_some_and(|value| value > 0)); + assert_eq!( + Some(1), + lookup_metric(&metrics_recorder, crate::db_stats::READER_REPLAY_MEMTABLES) + ); + assert_eq!( + Some(1), + lookup_metric( + &metrics_recorder, + crate::db_stats::READER_ACTIVE_CHECKPOINTS + ) + ); + + reader.close().await.unwrap(); + assert_eq!( + Some(0), + lookup_metric( + &metrics_recorder, + crate::db_stats::READER_ACTIVE_CHECKPOINTS + ) + ); + db.close().await.unwrap(); + } + + #[tokio::test] + async fn fixed_checkpoint_reader_should_not_manage_or_delete_its_checkpoint() { + use slatedb_common::metrics::{lookup_metric, DefaultMetricsRecorder}; + + let object_store: Arc = Arc::new(InMemory::new()); + let path = Path::from("/tmp/test_fixed_db_reader_checkpoint_metric"); + let db = Db::builder(path.clone(), Arc::clone(&object_store)) + .with_settings(Settings::default()) + .build() + .await + .unwrap(); + db.put(b"key", b"value").await.unwrap(); + let checkpoint = db + .create_checkpoint(CheckpointScope::All, &CheckpointOptions::default()) + .await + .unwrap(); + db.close().await.unwrap(); + + let metrics_recorder = Arc::new(DefaultMetricsRecorder::new()); + let reader = DbReader::builder(path.clone(), Arc::clone(&object_store)) + .with_reader_mode(DbReaderMode::Checkpoint(checkpoint.id)) + .with_metrics_recorder(metrics_recorder.clone()) + .build() + .await + .unwrap(); + + assert_eq!( + Some(0), + lookup_metric( + &metrics_recorder, + crate::db_stats::READER_ACTIVE_CHECKPOINTS + ) + ); + assert_eq!( + reader.get(b"key").await.unwrap(), + Some(Bytes::from("value")) + ); + + reader.close().await.unwrap(); + assert_eq!(reader.status().close_reason, Some(CloseReason::Clean)); + assert!(reader.get(b"key").await.is_err()); + assert_eq!( + Some(0), + lookup_metric( + &metrics_recorder, + crate::db_stats::READER_ACTIVE_CHECKPOINTS + ) + ); + + let manifest_store = ManifestStore::new(&path, Arc::clone(&object_store)); + let manifest = manifest_store.read_latest_manifest().await.unwrap(); + assert!(manifest + .manifest + .core + .find_checkpoint(checkpoint.id) + .is_some()); + + let drop_metrics_recorder = Arc::new(DefaultMetricsRecorder::new()); + let dropped_reader = DbReader::builder(path, Arc::clone(&object_store)) + .with_reader_mode(DbReaderMode::Checkpoint(checkpoint.id)) + .with_metrics_recorder(drop_metrics_recorder.clone()) + .build() + .await + .unwrap(); + assert_eq!( + Some(0), + lookup_metric( + &drop_metrics_recorder, + crate::db_stats::READER_ACTIVE_CHECKPOINTS + ) + ); + drop(dropped_reader); + assert_eq!( + Some(0), + lookup_metric( + &drop_metrics_recorder, + crate::db_stats::READER_ACTIVE_CHECKPOINTS + ) + ); + + let manifest = manifest_store.read_latest_manifest().await.unwrap(); + assert!(manifest + .manifest + .core + .find_checkpoint(checkpoint.id) + .is_some()); } impl TestProvider { @@ -3847,7 +5246,18 @@ mod tests { let timeout = Duration::from_secs(30); let start = tokio::time::Instant::now(); loop { - if reader.inner.state.read().manifest.core.tree.l0.len() == 1 { + if reader + .inner + .state + .read() + .generation + .manifest + .core + .tree + .l0 + .len() + == 1 + { break; } // The reader poller may observe the pre-flush manifest on one tick and diff --git a/slatedb/src/db_snapshot.rs b/slatedb/src/db_snapshot.rs index 102f005d74..30dac5d8c3 100644 --- a/slatedb/src/db_snapshot.rs +++ b/slatedb/src/db_snapshot.rs @@ -8,13 +8,30 @@ use crate::db_iter::DbIterator; use crate::types::KeyValue; use crate::db::DbInner; +use crate::db_reader::{DbReaderInner, ReaderState}; use crate::reader::ScanContext; use crate::DbReadOps; +/// An immutable snapshot-isolation view created by either [`crate::Db`] or +/// [`crate::DbReader`]. +/// +/// Reader-backed snapshots pin the reader's already-replayed local state. They +/// do not replay WALs or create a checkpoint per snapshot; snapshots from the +/// same manifest generation share the reader's GC checkpoint. pub struct DbSnapshot { - snapshot_id: Uuid, started_seq: u64, - db_inner: Arc, + backend: DbSnapshotBackend, +} + +enum DbSnapshotBackend { + Db { + snapshot_id: Uuid, + inner: Arc, + }, + Reader { + inner: Arc, + state: Arc, + }, } impl DbSnapshot { @@ -22,9 +39,18 @@ impl DbSnapshot { let (snapshot_id, started_seq) = db_inner.snapshot_manager.new_snapshot(seq); Arc::new(Self { - snapshot_id, started_seq, - db_inner, + backend: DbSnapshotBackend::Db { + snapshot_id, + inner: db_inner, + }, + }) + } + + pub(crate) fn new_reader(inner: Arc, state: Arc) -> Arc { + Arc::new(Self { + started_seq: state.applied_seq(), + backend: DbSnapshotBackend::Reader { inner, state }, }) } @@ -78,15 +104,32 @@ impl DbSnapshot { key: K, options: &ReadOptions, ) -> Result, crate::Error> { - self.db_inner.check_closed()?; - let db_state = self.db_inner.state.read().view(); - let kv = self - .db_inner - .reader - .get_key_value_with_options(key, options, &db_state, None, Some(self.started_seq)) - .await - .map_err(crate::Error::from)?; - Ok(kv) + match &self.backend { + DbSnapshotBackend::Db { inner, .. } => { + inner.check_closed()?; + let db_state = inner.state.read().view(); + inner + .reader + .get_key_value_with_options( + key, + options, + &db_state, + None, + Some(self.started_seq), + ) + .await + .map_err(crate::Error::from) + } + DbSnapshotBackend::Reader { inner, state } => inner + .snapshot_get_key_value_with_options( + Arc::clone(state), + self.started_seq, + key, + options, + ) + .await + .map_err(crate::Error::from), + } } /// Scan a range of keys using the default scan options. @@ -186,22 +229,36 @@ impl DbSnapshot { options: &ScanOptions, prefix: Option, ) -> Result { - self.db_inner.check_closed()?; - let db_state = self.db_inner.state.read().view(); - self.db_inner - .reader - .scan_with_options( - range, - options, - ScanContext { - db_state: &db_state, - write_batch_iter: None, - max_seq: Some(self.started_seq), + match &self.backend { + DbSnapshotBackend::Db { inner, .. } => { + inner.check_closed()?; + let db_state = inner.state.read().view(); + inner + .reader + .scan_with_options( + range, + options, + ScanContext { + db_state: &db_state, + write_batch_iter: None, + max_seq: Some(self.started_seq), + prefix, + }, + ) + .await + .map_err(Into::into) + } + DbSnapshotBackend::Reader { inner, state } => inner + .snapshot_scan_with_options( + Arc::clone(state), + self.started_seq, + range, + options, prefix, - }, - ) - .await - .map_err(Into::into) + ) + .await + .map_err(Into::into), + } } } @@ -250,9 +307,9 @@ impl DbReadOps for DbSnapshot { impl Drop for DbSnapshot { fn drop(&mut self) { - self.db_inner - .snapshot_manager - .drop_snapshot(&self.snapshot_id); + if let DbSnapshotBackend::Db { snapshot_id, inner } = &self.backend { + inner.snapshot_manager.drop_snapshot(snapshot_id); + } } } diff --git a/slatedb/src/db_state.rs b/slatedb/src/db_state.rs index 0731d4cda8..96f5613d9c 100644 --- a/slatedb/src/db_state.rs +++ b/slatedb/src/db_state.rs @@ -703,8 +703,8 @@ impl DbStateReader for DbStateView { Arc::clone(&self.memtable) } - fn imm_memtable(&self) -> &VecDeque> { - &self.state.imm_memtable + fn imm_memtables(&self) -> Box> + '_> { + Box::new(self.state.imm_memtable.iter().cloned()) } fn core(&self) -> &ManifestCore { @@ -734,7 +734,7 @@ pub(crate) fn collect_touched_segments( return std::collections::BTreeSet::new(); } let mut set = reader.memtable().touched_segments(); - for imm in reader.imm_memtable() { + for imm in reader.imm_memtables() { set.extend(imm.table().touched_segments()); } set diff --git a/slatedb/src/db_stats.rs b/slatedb/src/db_stats.rs index 89a40c7c60..d615044c6c 100644 --- a/slatedb/src/db_stats.rs +++ b/slatedb/src/db_stats.rs @@ -40,6 +40,12 @@ pub const SST_FILTER_NEGATIVE_COUNT: &str = db_stat_name!("sst_filter_negative_c /// write_amp = (`WAL_FLUSH_BYTES` + `L0_FLUSH_BYTES` + `compactor::stats::BYTES_COMPACTED`) /// / `MEMTABLE_WRITE_BYTES` pub const MEMTABLE_WRITE_BYTES: &str = db_stat_name!("memtable_write_bytes"); +pub const READER_WAL_REPLAY_SSTS: &str = db_stat_name!("reader_wal_replay_ssts"); +pub const READER_WAL_REPLAY_BYTES: &str = db_stat_name!("reader_wal_replay_bytes"); +pub const READER_WAL_REPLAY_BATCHES: &str = db_stat_name!("reader_wal_replay_batches"); +pub const READER_REPLAY_MEMTABLES: &str = db_stat_name!("reader_replay_memtables"); +pub const READER_ACTIVE_CHECKPOINTS: &str = db_stat_name!("reader_active_checkpoints"); +pub const READER_MANIFEST_POLLS: &str = db_stat_name!("reader_manifest_polls"); /// Label key distinguishing filter metrics for point lookups from those for /// prefix scans. Value is one of [`FILTER_KIND_POINT`] or @@ -75,6 +81,12 @@ pub(crate) struct DbStatsInner { pub(crate) merge_operator_read_operands: Arc, pub(crate) merge_operator_flush_operands: Arc, pub(crate) memtable_write_bytes: Arc, + pub(crate) reader_wal_replay_ssts: Arc, + pub(crate) reader_wal_replay_bytes: Arc, + pub(crate) reader_wal_replay_batches: Arc, + pub(crate) reader_replay_memtables: Arc, + pub(crate) reader_active_checkpoints: Arc, + pub(crate) reader_manifest_polls: Arc, } #[derive(Clone)] @@ -161,6 +173,20 @@ impl DbStats { .description(MERGE_OPERATOR_OPERANDS_DESCRIPTION) .register(), memtable_write_bytes: recorder.counter(MEMTABLE_WRITE_BYTES).register(), + reader_wal_replay_ssts: recorder.counter(READER_WAL_REPLAY_SSTS).register(), + reader_wal_replay_bytes: recorder.counter(READER_WAL_REPLAY_BYTES).register(), + reader_wal_replay_batches: recorder.counter(READER_WAL_REPLAY_BATCHES).register(), + reader_replay_memtables: recorder.gauge(READER_REPLAY_MEMTABLES).register(), + reader_active_checkpoints: recorder + .gauge(READER_ACTIVE_CHECKPOINTS) + .description( + "Number of GC checkpoints currently managed by this DbReader's manifest poller", + ) + .register(), + reader_manifest_polls: recorder + .counter(READER_MANIFEST_POLLS) + .description("Number of successful DbReader manifest polls completed") + .register(), }; DbStats { inner: Arc::new(inner), diff --git a/slatedb/src/error.rs b/slatedb/src/error.rs index c9bedaa5c2..fc88d3ff13 100644 --- a/slatedb/src/error.rs +++ b/slatedb/src/error.rs @@ -191,6 +191,15 @@ pub(crate) enum SlateDBError { #[error("checkpoint missing. checkpoint_id=`{0}`")] CheckpointMissing(Uuid), + #[error("checkpoint already exists. checkpoint_id=`{0}`")] + CheckpointAlreadyExists(Uuid), + + #[error("reader checkpoint lease lost. checkpoint_id=`{0}`")] + CheckpointLeaseLost(Uuid), + + #[error("reader snapshots are unsupported in FollowLatest mode")] + DbReaderSnapshotUnsupportedInFollowLatest, + #[error( "unsupported {format_name} format version. supported_versions=`{supported_versions:?}`, actual_version=`{actual_version}`" )] @@ -653,6 +662,7 @@ impl From for Error { SlateDBError::FoyerError(err) => Error::unavailable(msg).with_source(Box::new(err)), SlateDBError::TransactionalObjectTimeout { .. } => Error::unavailable(msg), SlateDBError::WalUnavailable(src) => Error::unavailable(msg).with_source(Box::new(src)), + SlateDBError::CheckpointLeaseLost(_) => Error::unavailable(msg), // Invalid errors SlateDBError::InvalidCachePartSize => Error::invalid(msg), @@ -671,6 +681,7 @@ impl From for Error { SlateDBError::InvalidCheckpointLifetime(_) => Error::invalid(msg), SlateDBError::InvalidManifestPollInterval(_) => Error::invalid(msg), SlateDBError::CheckpointLifetimeTooShort { .. } => Error::invalid(msg), + SlateDBError::DbReaderSnapshotUnsupportedInFollowLatest => Error::invalid(msg), SlateDBError::SeekKeyOutOfRange { .. } => Error::invalid(msg), SlateDBError::SeekKeyLessThanLastReturnedKey => Error::invalid(msg), SlateDBError::IdenticalClonePaths { .. } => Error::invalid(msg), @@ -712,6 +723,7 @@ impl From for Error { SlateDBError::BlockTransformError => Error::data(msg), SlateDBError::InvalidRowFlags { .. } => Error::data(msg), SlateDBError::CheckpointMissing(_) => Error::data(msg), + SlateDBError::CheckpointAlreadyExists(_) => Error::data(msg), SlateDBError::InvalidVersion { .. } => Error::data(msg), SlateDBError::ManifestMissing(_) => Error::data(msg), LatestTransactionalObjectVersionMissing => Error::data(msg), diff --git a/slatedb/src/manifest/store.rs b/slatedb/src/manifest/store.rs index 79a491a547..d4f1d9af03 100644 --- a/slatedb/src/manifest/store.rs +++ b/slatedb/src/manifest/store.rs @@ -2,7 +2,8 @@ use crate::checkpoint::Checkpoint; use crate::config::CheckpointOptions; use crate::error::SlateDBError; use crate::error::SlateDBError::{ - CheckpointMissing, InvalidDBState, LatestTransactionalObjectVersionMissing, ManifestMissing, + CheckpointAlreadyExists, CheckpointMissing, InvalidDBState, + LatestTransactionalObjectVersionMissing, ManifestMissing, }; use crate::flatbuffer_types::FlatBufferManifestCodec; use crate::manifest::{Manifest, ManifestCore, VersionedManifest}; @@ -16,7 +17,7 @@ use slatedb_txn_obj::{ DirtyObject, FenceableTransactionalObject, MonotonicId, SequencedStorageProtocol, SimpleTransactionalObject, TransactionalObject, TransactionalStorageProtocol, }; -use std::collections::BTreeMap; +use std::collections::{BTreeMap, BTreeSet}; use std::ops::RangeBounds; use std::sync::Arc; use std::time::Duration; @@ -301,6 +302,14 @@ impl StoredManifest { self.inner .maybe_apply_update(|sr| { let mut new_val = sr.object().clone(); + if new_val + .core + .checkpoints + .iter() + .any(|checkpoint| checkpoint.id == checkpoint_id) + { + return Err(CheckpointAlreadyExists(checkpoint_id)); + } let checkpoint = Self::new_checkpoint( &new_val, sr.id().into(), @@ -346,9 +355,42 @@ impl StoredManifest { .await?) } + /// Deletes several checkpoints in one manifest append. Missing checkpoint + /// IDs are treated as already deleted so cleanup remains idempotent. + pub(crate) async fn delete_checkpoints( + &mut self, + checkpoint_ids: &[Uuid], + ) -> Result<(), SlateDBError> { + if checkpoint_ids.is_empty() { + return Ok(()); + } + let checkpoint_ids = checkpoint_ids.iter().copied().collect::>(); + Ok(self + .inner + .maybe_apply_update(|sr| { + let mut new_val = sr.object().clone(); + let before = new_val.core.checkpoints.len(); + new_val + .core + .checkpoints + .retain(|cp| !checkpoint_ids.contains(&cp.id)); + let result: Result>, SlateDBError> = + if new_val.core.checkpoints.len() == before { + Ok(None) + } else { + let mut dirty = sr.prepare_dirty()?; + dirty.value = new_val; + Ok(Some(dirty)) + }; + result + }) + .await?) + } + /// Replace an existing checkpoint with a new checkpoint. If the old checkpoint /// is missing, the new checkpoint will still be added. This helps avoid /// issuing two manifest updates when creating a new checkpoint. + #[cfg(test)] pub(crate) async fn replace_checkpoint( &mut self, old_checkpoint_id: Uuid, @@ -386,6 +428,7 @@ impl StoredManifest { Ok(new_checkpoint) } + #[cfg(test)] pub(crate) async fn refresh_checkpoint( &mut self, checkpoint_id: Uuid, @@ -417,6 +460,47 @@ impl StoredManifest { Ok(checkpoint) } + /// Refreshes several checkpoint leases in one manifest append. + pub(crate) async fn refresh_checkpoints( + &mut self, + checkpoint_ids: &[Uuid], + new_lifetime: Duration, + ) -> Result, SlateDBError> { + if checkpoint_ids.is_empty() { + return Ok(Vec::new()); + } + let checkpoint_ids = checkpoint_ids.iter().copied().collect::>(); + let clock = Arc::clone(&self.clock); + self.inner + .maybe_apply_update(|sr| { + let mut new_val = sr.object().clone(); + for checkpoint_id in &checkpoint_ids { + if !new_val + .core + .checkpoints + .iter() + .any(|checkpoint| checkpoint.id == *checkpoint_id) + { + return Err(CheckpointMissing(*checkpoint_id)); + } + } + let expire_time = clock.now() + new_lifetime; + for checkpoint in &mut new_val.core.checkpoints { + if checkpoint_ids.contains(&checkpoint.id) { + checkpoint.expire_time = Some(expire_time); + } + } + let mut dirty = sr.prepare_dirty()?; + dirty.value = new_val; + Ok(Some(dirty)) + }) + .await?; + Ok(checkpoint_ids + .iter() + .filter_map(|id| self.db_state().find_checkpoint(*id).cloned()) + .collect()) + } + pub(crate) async fn update( &mut self, dirty: DirtyObject, @@ -1298,6 +1382,73 @@ mod tests { ); } + #[tokio::test] + async fn should_refresh_checkpoints_atomically_in_one_manifest() { + let ms = new_memory_manifest_store(); + let mut sm = StoredManifest::create_new_db( + ms, + ManifestCore::new(), + Arc::new(DefaultSystemClock::new()), + ) + .await + .unwrap(); + let options = CheckpointOptions { + lifetime: Some(Duration::from_secs(100)), + ..CheckpointOptions::default() + }; + let first = sm + .write_checkpoint(uuid::Uuid::new_v4(), &options) + .await + .unwrap(); + let second = sm + .write_checkpoint(uuid::Uuid::new_v4(), &options) + .await + .unwrap(); + let manifest_id = sm.id(); + + let refreshed = sm + .refresh_checkpoints(&[first.id, second.id], Duration::from_secs(500)) + .await + .unwrap(); + + assert_eq!(manifest_id + 1, sm.id()); + assert_eq!(2, refreshed.len()); + assert_eq!(refreshed[0].expire_time, refreshed[1].expire_time); + assert!(refreshed[0].expire_time > first.expire_time); + } + + #[tokio::test] + async fn should_not_partially_refresh_when_any_checkpoint_is_missing() { + let ms = new_memory_manifest_store(); + let mut sm = StoredManifest::create_new_db( + ms, + ManifestCore::new(), + Arc::new(DefaultSystemClock::new()), + ) + .await + .unwrap(); + let checkpoint = sm + .write_checkpoint(uuid::Uuid::new_v4(), &CheckpointOptions::default()) + .await + .unwrap(); + let manifest_id = sm.id(); + let missing = uuid::Uuid::new_v4(); + + let result = sm + .refresh_checkpoints(&[checkpoint.id, missing], Duration::from_secs(500)) + .await; + + assert!(matches!(result, Err(SlateDBError::CheckpointMissing(id)) if id == missing)); + assert_eq!(manifest_id, sm.id()); + assert_eq!( + checkpoint.expire_time, + sm.db_state() + .find_checkpoint(checkpoint.id) + .unwrap() + .expire_time + ); + } + #[tokio::test] async fn should_fail_refresh_if_checkpoint_missing() { let ms = new_memory_manifest_store(); @@ -1404,6 +1555,103 @@ mod tests { assert_eq!(None, sm.manifest().core.find_checkpoint(checkpoint.id)); } + #[tokio::test] + async fn should_delete_checkpoints_idempotently_in_one_manifest() { + let ms = new_memory_manifest_store(); + let mut sm = StoredManifest::create_new_db( + ms, + ManifestCore::new(), + Arc::new(DefaultSystemClock::new()), + ) + .await + .unwrap(); + let first = sm + .write_checkpoint(uuid::Uuid::new_v4(), &CheckpointOptions::default()) + .await + .unwrap(); + let second = sm + .write_checkpoint(uuid::Uuid::new_v4(), &CheckpointOptions::default()) + .await + .unwrap(); + let manifest_id = sm.id(); + + sm.delete_checkpoints(&[first.id, second.id]).await.unwrap(); + + assert_eq!(manifest_id + 1, sm.id()); + assert!(sm.db_state().checkpoints.is_empty()); + let manifest_id = sm.id(); + sm.delete_checkpoints(&[first.id, second.id]).await.unwrap(); + assert_eq!(manifest_id, sm.id()); + } + + #[tokio::test] + async fn should_reject_duplicate_checkpoint_ids_without_appending() { + let ms = new_memory_manifest_store(); + let mut sm = StoredManifest::create_new_db( + ms, + ManifestCore::new(), + Arc::new(DefaultSystemClock::new()), + ) + .await + .unwrap(); + let checkpoint_id = uuid::Uuid::new_v4(); + sm.write_checkpoint(checkpoint_id, &CheckpointOptions::default()) + .await + .unwrap(); + let manifest_id = sm.id(); + + let result = sm + .write_checkpoint(checkpoint_id, &CheckpointOptions::default()) + .await; + + assert!( + matches!(result, Err(SlateDBError::CheckpointAlreadyExists(id)) if id == checkpoint_id) + ); + assert_eq!(manifest_id, sm.id()); + assert_eq!(1, sm.db_state().checkpoints.len()); + } + + #[tokio::test] + async fn checkpoint_append_should_merge_with_a_concurrent_manifest_writer() { + let ms = new_memory_manifest_store(); + StoredManifest::create_new_db( + Arc::clone(&ms), + ManifestCore::new(), + Arc::new(DefaultSystemClock::new()), + ) + .await + .unwrap(); + let mut reader_manifest = + StoredManifest::load(Arc::clone(&ms), Arc::new(DefaultSystemClock::new())) + .await + .unwrap(); + let mut writer_manifest = + StoredManifest::load(Arc::clone(&ms), Arc::new(DefaultSystemClock::new())) + .await + .unwrap(); + let checkpoint_id = uuid::Uuid::new_v4(); + let checkpoint_options = CheckpointOptions::default(); + + let (checkpoint_result, writer_result) = tokio::join!( + reader_manifest.write_checkpoint(checkpoint_id, &checkpoint_options), + writer_manifest.maybe_apply_update(|manifest| { + let mut dirty = manifest.prepare_dirty()?; + dirty.value.core.last_l0_seq = 42; + Ok(Some(dirty)) + }) + ); + + checkpoint_result.unwrap(); + writer_result.unwrap(); + let latest = ms.read_latest_manifest().await.unwrap(); + assert_eq!(42, latest.manifest.core.last_l0_seq); + assert!(latest + .manifest + .core + .find_checkpoint(checkpoint_id) + .is_some()); + } + #[tokio::test] async fn should_ignore_missing_checkpoint_if_deleting() { let ms = new_memory_manifest_store(); diff --git a/slatedb/src/reader.rs b/slatedb/src/reader.rs index 9049e7e1aa..c66eb996ff 100644 --- a/slatedb/src/reader.rs +++ b/slatedb/src/reader.rs @@ -22,7 +22,9 @@ use std::sync::Arc; pub(crate) trait DbStateReader { fn memtable(&self) -> Arc; - fn imm_memtable(&self) -> &VecDeque>; + /// Returns immutable memtables newest-first. The iterator form permits + /// read-only replicas to use a structurally shared persistent chain. + fn imm_memtables(&self) -> Box> + '_>; fn core(&self) -> &ManifestCore; } @@ -141,7 +143,7 @@ impl Reader { ) -> Result { let mut memtables = VecDeque::new(); memtables.push_back(db_state.memtable()); - for memtable in db_state.imm_memtable() { + for memtable in db_state.imm_memtables() { memtables.push_back(memtable.table()); } let mem_iters = memtables @@ -379,7 +381,7 @@ impl Reader { .memtable() .range(range.clone(), sst_iter_options.order), )); - for memtable in db_state.imm_memtable() { + for memtable in db_state.imm_memtables() { all_iters.push(Box::new( memtable .table() @@ -605,8 +607,8 @@ mod tests { self.memtable.clone() } - fn imm_memtable(&self) -> &VecDeque> { - &self.imm_memtable + fn imm_memtables(&self) -> Box> + '_> { + Box::new(self.imm_memtable.iter().cloned()) } fn core(&self) -> &ManifestCore { From 7660fb69b82bcdffe6008255c757c962d60fe089 Mon Sep 17 00:00:00 2001 From: Matthew Sanetra <41018997+matthewsanetra@users.noreply.github.com> Date: Thu, 16 Jul 2026 18:13:09 +0100 Subject: [PATCH 52/65] Harden reader snapshot WAL replay test --- slatedb/src/db_reader.rs | 11 ++++++++++- 1 file changed, 10 insertions(+), 1 deletion(-) diff --git a/slatedb/src/db_reader.rs b/slatedb/src/db_reader.rs index 41947e060d..e7481fc570 100644 --- a/slatedb/src/db_reader.rs +++ b/slatedb/src/db_reader.rs @@ -2981,7 +2981,16 @@ mod tests { }) .await .unwrap(); - tokio::time::sleep(Duration::from_millis(20)).await; + tokio::time::timeout(Duration::from_secs(10), async { + loop { + if reader.get(b"key").await.unwrap() == Some(Bytes::from_static(b"v2")) { + break; + } + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await + .expect("live reader should replay the second WAL generation"); assert_eq!(reader.get(b"key").await.unwrap(), Some(Bytes::from("v2"))); assert_eq!(snapshot.get(b"key").await.unwrap(), Some(Bytes::from("v1"))); From 8b7c2639e53246a8d93e1fd34542550e444e19c3 Mon Sep 17 00:00:00 2001 From: xav-db Date: Mon, 6 Jul 2026 13:59:02 +0100 Subject: [PATCH 53/65] Add multi-get functionality across database components - Introduced `multi_get` and `multi_get_with_options` methods in `Db`, `DbReader`, `DbSnapshot`, and `DbTransaction` to retrieve multiple values efficiently. - Implemented a new `WriteBatchLookup` enum to facilitate lookup operations in `WriteBatch`. - Enhanced the `lookup_latest_for_key` method to support the new multi-get functionality. - Added tests to ensure correctness of multi-get operations, preserving input order and handling duplicates. This update improves the performance and usability of batch retrieval operations in the database. --- slatedb/src/batch.rs | 16 + slatedb/src/db.rs | 94 ++++++ slatedb/src/db_reader.rs | 105 ++++++- slatedb/src/db_snapshot.rs | 71 +++++ slatedb/src/db_state.rs | 2 +- slatedb/src/db_transaction.rs | 312 ++++++++++++++++++- slatedb/src/ops.rs | 48 +++ slatedb/src/reader.rs | 557 +++++++++++++++++++++++++++++++++- 8 files changed, 1198 insertions(+), 7 deletions(-) diff --git a/slatedb/src/batch.rs b/slatedb/src/batch.rs index 836e32b7c4..2be7bcdebd 100644 --- a/slatedb/src/batch.rs +++ b/slatedb/src/batch.rs @@ -69,6 +69,13 @@ pub(crate) enum WriteOp { Merge(Bytes, MergeOptions), } +pub(crate) enum WriteBatchLookup { + Put(Bytes), + Delete, + Merge, + NotPresent, +} + impl std::fmt::Debug for WriteOp { fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { fn trunc(bytes: &Bytes) -> String { @@ -278,6 +285,15 @@ impl WriteBatch { self.ops.is_empty() } + pub(crate) fn lookup_latest_for_key(&self, key: &[u8]) -> WriteBatchLookup { + match self.ops.get(key).and_then(|ops| ops.last()) { + Some(WriteOp::Put(value, _)) => WriteBatchLookup::Put(value.clone()), + Some(WriteOp::Delete) => WriteBatchLookup::Delete, + Some(WriteOp::Merge(_, _)) => WriteBatchLookup::Merge, + None => WriteBatchLookup::NotPresent, + } + } + pub(crate) fn has_merge_ops(&self) -> bool { self.has_merge_ops } diff --git a/slatedb/src/db.rs b/slatedb/src/db.rs index 57f73083cc..4c331139ea 100644 --- a/slatedb/src/db.rs +++ b/slatedb/src/db.rs @@ -230,6 +230,32 @@ impl DbInner { .await } + pub(crate) async fn multi_get_with_options>( + &self, + keys: &[K], + options: &ReadOptions, + ) -> Result>, SlateDBError> { + self.multi_get_key_value_with_options(keys, options) + .await + .map(|values| values.into_iter().map(|kv| kv.map(|kv| kv.value)).collect()) + } + + pub(crate) async fn multi_get_key_value_with_options>( + &self, + keys: &[K], + options: &ReadOptions, + ) -> Result>, SlateDBError> { + self.check_closed()?; + let db_state = self.state.read().view(); + let keys = keys + .iter() + .map(|key| Bytes::copy_from_slice(key.as_ref())) + .collect::>(); + self.reader + .multi_get_key_value_with_options(&keys, options, &db_state, None) + .await + } + /// Shared scan path for plain range scans and prefix scans. When /// `prefix` is set, every key in `range` starts with it and prefix /// bloom filters are consulted to skip non-matching SSTs. @@ -930,6 +956,31 @@ impl Db { Ok(kv) } + /// Get multiple values from the database with default read options. + /// + /// The returned vector preserves input order and duplicates. + pub async fn multi_get + Send + Sync>( + &self, + keys: &[K], + ) -> Result>, crate::Error> { + self.multi_get_with_options(keys, &ReadOptions::default()) + .await + } + + /// Get multiple values from the database with custom read options. + /// + /// The returned vector preserves input order and duplicates. + pub async fn multi_get_with_options + Send + Sync>( + &self, + keys: &[K], + options: &ReadOptions, + ) -> Result>, crate::Error> { + self.inner + .multi_get_with_options(keys, options) + .await + .map_err(crate::Error::from) + } + /// Scan a range of keys using the default scan options. /// /// returns a `DbIterator` @@ -1924,6 +1975,17 @@ impl DbReadOps for Db { Db::get_key_value_with_options(self, key, options).await } + async fn multi_get_with_options( + &self, + keys: &[K], + options: &ReadOptions, + ) -> Result>, crate::Error> + where + K: AsRef<[u8]> + Send + Sync, + { + Db::multi_get_with_options(self, keys, options).await + } + async fn scan_with_options( &self, range: T, @@ -2401,6 +2463,38 @@ mod tests { kv_store.close().await.unwrap(); } + #[tokio::test] + async fn test_multi_get_matches_repeated_get_with_duplicates() { + let object_store: Arc = Arc::new(InMemory::new()); + let kv_store = Db::builder("/tmp/test_multi_get", object_store) + .with_settings(test_db_options(0, 1024, None)) + .build() + .await + .unwrap(); + + kv_store.put(b"k1", b"v1").await.unwrap(); + kv_store.put(b"k2", b"v2").await.unwrap(); + kv_store.put(b"k3", b"v3").await.unwrap(); + kv_store.flush().await.unwrap(); + + let keys: [&[u8]; 6] = [b"k1", b"missing", b"k2", b"k1", b"k3", b"missing"]; + let multi = kv_store.multi_get(&keys).await.unwrap(); + let mut repeated = Vec::with_capacity(keys.len()); + for key in keys { + repeated.push(kv_store.get(key).await.unwrap()); + } + + assert_eq!(multi, repeated); + assert_eq!(multi[0], Some(Bytes::from_static(b"v1"))); + assert_eq!(multi[1], None); + assert_eq!(multi[2], Some(Bytes::from_static(b"v2"))); + assert_eq!(multi[3], Some(Bytes::from_static(b"v1"))); + assert_eq!(multi[4], Some(Bytes::from_static(b"v3"))); + assert_eq!(multi[5], None); + + kv_store.close().await.unwrap(); + } + #[tokio::test] async fn test_manifest_returns_current_versioned_manifest() { let object_store: Arc = Arc::new(InMemory::new()); diff --git a/slatedb/src/db_reader.rs b/slatedb/src/db_reader.rs index e7481fc570..2c0da334f0 100644 --- a/slatedb/src/db_reader.rs +++ b/slatedb/src/db_reader.rs @@ -361,7 +361,7 @@ impl DbStateReader for ReaderState { Arc::clone(&EMPTY_TABLE) } - fn imm_memtables(&self) -> Box> + '_> { + fn imm_memtables(&self) -> Box> + Send + '_> { Box::new(self.imm_memtable.iter()) } @@ -543,6 +543,33 @@ impl DbReaderInner { .await } + async fn multi_get_with_options + Send + Sync>( + &self, + keys: &[K], + options: &ReadOptions, + ) -> Result>, SlateDBError> { + self.multi_get_key_value_with_options(keys, options) + .await + .map(|values| values.into_iter().map(|kv| kv.map(|kv| kv.value)).collect()) + } + + async fn multi_get_key_value_with_options + Send + Sync>( + &self, + keys: &[K], + options: &ReadOptions, + ) -> Result>, SlateDBError> { + self.check_closed()?; + let db_state = Arc::clone(&self.state.read()); + let _permit = db_state.generation.acquire()?; + let keys = keys + .iter() + .map(|key| Bytes::copy_from_slice(key.as_ref())) + .collect::>(); + self.reader + .multi_get_key_value_with_options(&keys, options, db_state.as_ref(), None) + .await + } + pub(crate) async fn snapshot_get_key_value_with_options + Send>( &self, state: Arc, @@ -557,6 +584,26 @@ impl DbReaderInner { .await } + pub(crate) async fn snapshot_multi_get_key_value_with_options< + K: AsRef<[u8]> + Send + Sync, + >( + &self, + state: Arc, + max_seq: u64, + keys: &[K], + options: &ReadOptions, + ) -> Result>, SlateDBError> { + self.check_closed()?; + let _permit = state.generation.acquire()?; + let keys = keys + .iter() + .map(|key| Bytes::copy_from_slice(key.as_ref())) + .collect::>(); + self.reader + .multi_get_key_value_with_options(&keys, options, state.as_ref(), Some(max_seq)) + .await + } + async fn scan_with_options( &self, range: BytesRange, @@ -1635,6 +1682,31 @@ impl DbReader { Ok(kv) } + /// Get multiple values from the reader with default read options. + /// + /// The returned vector preserves input order and duplicates. + pub async fn multi_get + Send + Sync>( + &self, + keys: &[K], + ) -> Result>, crate::Error> { + self.multi_get_with_options(keys, &ReadOptions::default()) + .await + } + + /// Get multiple values from the reader with custom read options. + /// + /// The returned vector preserves input order and duplicates. + pub async fn multi_get_with_options + Send + Sync>( + &self, + keys: &[K], + options: &ReadOptions, + ) -> Result>, crate::Error> { + self.inner + .multi_get_with_options(keys, options) + .await + .map_err(Into::into) + } + /// Scan a range of keys using the default scan options. /// /// returns a `DbIterator` @@ -1880,6 +1952,17 @@ impl DbReadOps for DbReader { DbReader::get_key_value_with_options(self, key, options).await } + async fn multi_get_with_options( + &self, + keys: &[K], + options: &ReadOptions, + ) -> Result>, crate::Error> + where + K: AsRef<[u8]> + Send + Sync, + { + DbReader::multi_get_with_options(self, keys, options).await + } + async fn scan_with_options( &self, range: T, @@ -2976,6 +3059,9 @@ mod tests { db.put_with_options(b"key", b"v2", &PutOptions::default(), &write_options) .await .unwrap(); + db.put_with_options(b"new", b"v3", &PutOptions::default(), &write_options) + .await + .unwrap(); db.flush_with_options(FlushOptions { flush_type: FlushType::Wal, }) @@ -2994,6 +3080,23 @@ mod tests { assert_eq!(reader.get(b"key").await.unwrap(), Some(Bytes::from("v2"))); assert_eq!(snapshot.get(b"key").await.unwrap(), Some(Bytes::from("v1"))); + let keys = [&b"key"[..], &b"key"[..], &b"new"[..]]; + assert_eq!( + snapshot.multi_get(&keys).await.unwrap(), + vec![ + Some(Bytes::from_static(b"v1")), + Some(Bytes::from_static(b"v1")), + None, + ] + ); + assert_eq!( + reader.multi_get(&keys).await.unwrap(), + vec![ + Some(Bytes::from_static(b"v2")), + Some(Bytes::from_static(b"v2")), + Some(Bytes::from_static(b"v3")), + ] + ); assert_eq!( reader.snapshot().await.unwrap().get(b"key").await.unwrap(), Some(Bytes::from("v2")) diff --git a/slatedb/src/db_snapshot.rs b/slatedb/src/db_snapshot.rs index 30dac5d8c3..d6dbee3ebe 100644 --- a/slatedb/src/db_snapshot.rs +++ b/slatedb/src/db_snapshot.rs @@ -132,6 +132,66 @@ impl DbSnapshot { } } + /// Get multiple values from the snapshot with default read options. + /// + /// The returned vector preserves input order and duplicates. + pub async fn multi_get + Send + Sync>( + &self, + keys: &[K], + ) -> Result>, crate::Error> { + self.multi_get_with_options(keys, &ReadOptions::default()) + .await + } + + /// Get multiple values from the snapshot with custom read options. + /// + /// The returned vector preserves input order and duplicates. + pub async fn multi_get_with_options + Send + Sync>( + &self, + keys: &[K], + options: &ReadOptions, + ) -> Result>, crate::Error> { + self.multi_get_key_value_with_options(keys, options) + .await + .map(|values| values.into_iter().map(|kv| kv.map(|kv| kv.value)).collect()) + } + + async fn multi_get_key_value_with_options + Send + Sync>( + &self, + keys: &[K], + options: &ReadOptions, + ) -> Result>, crate::Error> { + match &self.backend { + DbSnapshotBackend::Db { inner, .. } => { + inner.check_closed()?; + let db_state = inner.state.read().view(); + let keys = keys + .iter() + .map(|key| Bytes::copy_from_slice(key.as_ref())) + .collect::>(); + inner + .reader + .multi_get_key_value_with_options( + &keys, + options, + &db_state, + Some(self.started_seq), + ) + .await + .map_err(crate::Error::from) + } + DbSnapshotBackend::Reader { inner, state } => inner + .snapshot_multi_get_key_value_with_options( + Arc::clone(state), + self.started_seq, + keys, + options, + ) + .await + .map_err(crate::Error::from), + } + } + /// Scan a range of keys using the default scan options. /// /// ## Arguments @@ -280,6 +340,17 @@ impl DbReadOps for DbSnapshot { DbSnapshot::get_key_value_with_options(self, key, options).await } + async fn multi_get_with_options( + &self, + keys: &[K], + options: &ReadOptions, + ) -> Result>, crate::Error> + where + K: AsRef<[u8]> + Send + Sync, + { + DbSnapshot::multi_get_with_options(self, keys, options).await + } + async fn scan_with_options( &self, range: T, diff --git a/slatedb/src/db_state.rs b/slatedb/src/db_state.rs index 96f5613d9c..13bc49915f 100644 --- a/slatedb/src/db_state.rs +++ b/slatedb/src/db_state.rs @@ -703,7 +703,7 @@ impl DbStateReader for DbStateView { Arc::clone(&self.memtable) } - fn imm_memtables(&self) -> Box> + '_> { + fn imm_memtables(&self) -> Box> + Send + '_> { Box::new(self.state.imm_memtable.iter().cloned()) } diff --git a/slatedb/src/db_transaction.rs b/slatedb/src/db_transaction.rs index 5d9c1fdf02..34855d333b 100644 --- a/slatedb/src/db_transaction.rs +++ b/slatedb/src/db_transaction.rs @@ -4,7 +4,7 @@ use std::collections::HashSet; use std::sync::Arc; use uuid::Uuid; -use crate::batch::{WriteBatch, WriteBatchIterator}; +use crate::batch::{WriteBatch, WriteBatchIterator, WriteBatchLookup}; use crate::bytes_range::{ByteRangeBounds, BytesRange}; use crate::config::{MergeOptions, PutOptions, ReadOptions, ScanOptions, WriteOptions}; use crate::db::DbInner; @@ -189,6 +189,148 @@ impl DbTransaction { Ok(kv) } + /// Get multiple values from the transaction with default read options. + /// This operation tracks all read keys for conflict detection in SSI mode. + pub async fn multi_get + Send + Sync>( + &self, + keys: &[K], + ) -> Result>, crate::Error> { + self.multi_get_with_options(keys, &ReadOptions::default()) + .await + } + + /// Get multiple values from the transaction with custom read options. + /// This operation tracks all read keys for conflict detection in SSI mode. + pub async fn multi_get_with_options + Send + Sync>( + &self, + keys: &[K], + options: &ReadOptions, + ) -> Result>, crate::Error> { + self.multi_get_key_value_with_options(keys, options) + .await + .map(|values| values.into_iter().map(|kv| kv.map(|kv| kv.value)).collect()) + } + + async fn multi_get_key_value_with_options + Send + Sync>( + &self, + keys: &[K], + options: &ReadOptions, + ) -> Result>, crate::Error> { + self.db_inner.check_closed()?; + + if self.isolation_level == IsolationLevel::SerializableSnapshot { + let read_keys = keys + .iter() + .map(|key| Bytes::copy_from_slice(key.as_ref())) + .collect::>(); + self.txn_manager.track_read_keys(&self.txn_id, read_keys); + } + + let db_state = self.db_inner.state.read().view(); + + let mut key_to_idx = std::collections::HashMap::::with_capacity(keys.len()); + let mut unique_keys = Vec::::with_capacity(keys.len()); + let mut output_positions = Vec::>::with_capacity(keys.len()); + + for (output_idx, key) in keys.iter().enumerate() { + let key = Bytes::copy_from_slice(key.as_ref()); + if let Some(existing_idx) = key_to_idx.get(&key).copied() { + output_positions[existing_idx].push(output_idx); + continue; + } + + let key_idx = unique_keys.len(); + key_to_idx.insert(key.clone(), key_idx); + unique_keys.push(key); + output_positions.push(vec![output_idx]); + } + + let mut resolved = vec![false; unique_keys.len()]; + let mut values = vec![None; unique_keys.len()]; + let mut fallback_to_point_get = vec![false; unique_keys.len()]; + + { + let write_batch = self.write_batch.read(); + for (key_idx, key) in unique_keys.iter().enumerate() { + match write_batch.lookup_latest_for_key(key.as_ref()) { + WriteBatchLookup::Put(value) => { + resolved[key_idx] = true; + values[key_idx] = Some(KeyValue { + key: key.clone(), + value, + seq: u64::MAX, + create_ts: 0, + expire_ts: None, + }); + } + WriteBatchLookup::Delete => { + resolved[key_idx] = true; + } + WriteBatchLookup::Merge => { + fallback_to_point_get[key_idx] = true; + } + WriteBatchLookup::NotPresent => {} + } + } + } + + let reader_key_indices = unique_keys + .iter() + .enumerate() + .filter_map(|(idx, _)| (!resolved[idx] && !fallback_to_point_get[idx]).then_some(idx)) + .collect::>(); + if !reader_key_indices.is_empty() { + let reader_keys = reader_key_indices + .iter() + .map(|idx| unique_keys[*idx].clone()) + .collect::>(); + let reader_values = self + .db_inner + .reader + .multi_get_key_value_with_options( + &reader_keys, + options, + &db_state, + Some(self.started_seq), + ) + .await + .map_err(crate::Error::from)?; + for (key_idx, value) in reader_key_indices + .into_iter() + .zip(reader_values.into_iter()) + { + resolved[key_idx] = true; + values[key_idx] = value; + } + } + + for key_idx in fallback_to_point_get + .iter() + .enumerate() + .filter_map(|(idx, should_fallback)| should_fallback.then_some(idx)) + { + values[key_idx] = self + .get_key_value_with_options(unique_keys[key_idx].as_ref(), options) + .await?; + resolved[key_idx] = true; + } + + let mut result = vec![None; keys.len()]; + for (key_idx, positions) in output_positions.into_iter().enumerate() { + let value = if resolved[key_idx] { + values[key_idx].clone() + } else { + None + }; + + for position in positions { + result[position] = value.clone(); + } + } + + Ok(result) + } + /// Scan a range of keys using the default scan options. /// This operation will track the read range for conflict detection in SSI mode. /// @@ -380,6 +522,25 @@ impl DbTransaction { Ok(()) } + /// Put an owned key-value pair into the transaction without copying the + /// value bytes into the transaction write batch. + pub fn put_bytes(&self, key: Bytes, value: Bytes) -> Result<(), crate::Error> { + self.put_bytes_with_options(key, value, &PutOptions::default()) + } + + /// Put an owned key-value pair into the transaction with custom options. + pub fn put_bytes_with_options( + &self, + key: Bytes, + value: Bytes, + options: &PutOptions, + ) -> Result<(), crate::Error> { + self.write_batch + .write() + .put_bytes_with_options(key, value, options); + Ok(()) + } + /// Mark keys as read for conflict detection. /// /// This method explicitly tracks read operations for conflict detection. When keys are @@ -658,6 +819,17 @@ impl DbReadOps for DbTransaction { DbTransaction::get_key_value_with_options(self, key, options).await } + async fn multi_get_with_options( + &self, + keys: &[K], + options: &ReadOptions, + ) -> Result>, crate::Error> + where + K: AsRef<[u8]> + Send + Sync, + { + DbTransaction::multi_get_with_options(self, keys, options).await + } + async fn scan_with_options( &self, range: T, @@ -698,6 +870,15 @@ impl DbTransactionOps for DbTransaction { DbTransaction::put_with_options(self, key, value, options) } + fn put_bytes_with_options( + &self, + key: Bytes, + value: Bytes, + options: &PutOptions, + ) -> Result<(), crate::Error> { + DbTransaction::put_bytes_with_options(self, key, value, options) + } + fn delete>(&self, key: K) -> Result<(), crate::Error> { DbTransaction::delete(self, key) } @@ -855,6 +1036,135 @@ mod tests { txn.commit().await.unwrap(); } + #[tokio::test] + async fn test_txn_put_bytes_read_your_writes() { + let object_store: Arc = Arc::new(InMemory::new()); + let db = crate::Db::open("test_txn_put_bytes", object_store) + .await + .unwrap(); + + let txn = db.begin(IsolationLevel::Snapshot).await.unwrap(); + txn.put_bytes( + Bytes::from_static(b"bytes_key"), + Bytes::from_static(b"bytes_value"), + ) + .unwrap(); + + assert_eq!( + txn.get(b"bytes_key").await.unwrap(), + Some(Bytes::from_static(b"bytes_value")) + ); + txn.commit().await.unwrap(); + + assert_eq!( + db.get(b"bytes_key").await.unwrap(), + Some(Bytes::from_static(b"bytes_value")) + ); + } + + #[tokio::test] + async fn test_txn_multi_get_read_your_writes() { + let object_store: Arc = Arc::new(InMemory::new()); + let db = crate::Db::open("test_txn_multi_get_read_your_writes", object_store) + .await + .unwrap(); + + db.put(b"k1", b"db_v1").await.unwrap(); + + let txn = db + .begin(IsolationLevel::SerializableSnapshot) + .await + .unwrap(); + txn.put(b"k1", b"txn_v1").unwrap(); + txn.put_bytes(Bytes::from_static(b"k2"), Bytes::from_static(b"txn_v2")) + .unwrap(); + + let keys: [&[u8]; 4] = [b"k1", b"k2", b"missing", b"k1"]; + let values = txn.multi_get(&keys).await.unwrap(); + + assert_eq!(values[0], Some(Bytes::from_static(b"txn_v1"))); + assert_eq!(values[1], Some(Bytes::from_static(b"txn_v2"))); + assert_eq!(values[2], None); + assert_eq!(values[3], Some(Bytes::from_static(b"txn_v1"))); + + txn.commit().await.unwrap(); + } + + #[tokio::test] + async fn test_txn_multi_get_matches_get_for_pending_put_delete_and_isolation() { + let object_store: Arc = Arc::new(InMemory::new()); + let db = crate::Db::open("test_txn_multi_get_pending_overlay", object_store) + .await + .unwrap(); + + db.put(b"k1", b"db_v1").await.unwrap(); + db.put(b"k2", b"db_v2").await.unwrap(); + db.put(b"k3", b"db_v3").await.unwrap(); + + let txn = db.begin(IsolationLevel::Snapshot).await.unwrap(); + txn.put(b"k1", b"txn_v1").unwrap(); + txn.delete(b"k2").unwrap(); + txn.put(b"k4", b"txn_v4").unwrap(); + + let keys: [&[u8]; 7] = [b"k1", b"k2", b"k3", b"k4", b"missing", b"k1", b"k2"]; + let multi = txn.multi_get(&keys).await.unwrap(); + let mut single = Vec::with_capacity(keys.len()); + for key in keys { + single.push(txn.get(key).await.unwrap()); + } + + assert_eq!(multi, single); + assert_eq!(multi[0], Some(Bytes::from_static(b"txn_v1"))); + assert_eq!(multi[1], None); + assert_eq!(multi[2], Some(Bytes::from_static(b"db_v3"))); + assert_eq!(multi[3], Some(Bytes::from_static(b"txn_v4"))); + assert_eq!(multi[4], None); + assert_eq!(multi[5], Some(Bytes::from_static(b"txn_v1"))); + assert_eq!(multi[6], None); + + let concurrent = db.begin(IsolationLevel::Snapshot).await.unwrap(); + let concurrent_values = concurrent + .multi_get(&[b"k1".as_ref(), b"k2".as_ref(), b"k4".as_ref()]) + .await + .unwrap(); + assert_eq!(concurrent_values[0], Some(Bytes::from_static(b"db_v1"))); + assert_eq!(concurrent_values[1], Some(Bytes::from_static(b"db_v2"))); + assert_eq!(concurrent_values[2], None); + concurrent.commit().await.unwrap(); + + txn.commit().await.unwrap(); + } + + #[tokio::test] + async fn test_txn_multi_get_tracks_reads_for_serializable_conflicts() { + let object_store: Arc = Arc::new(InMemory::new()); + let db = crate::Db::open("test_txn_multi_get_conflicts", object_store) + .await + .unwrap(); + + db.put(b"k1", b"db_v1").await.unwrap(); + + let txn1 = db + .begin(IsolationLevel::SerializableSnapshot) + .await + .unwrap(); + let keys: [&[u8]; 3] = [b"k1", b"k1", b"missing"]; + let values = txn1.multi_get(&keys).await.unwrap(); + assert_eq!(values[0], Some(Bytes::from_static(b"db_v1"))); + assert_eq!(values[1], Some(Bytes::from_static(b"db_v1"))); + assert_eq!(values[2], None); + + let txn2 = db + .begin(IsolationLevel::SerializableSnapshot) + .await + .unwrap(); + txn2.put(b"k1", b"db_v2").unwrap(); + txn2.commit().await.unwrap(); + + txn1.put(b"k2", b"txn1_write").unwrap(); + assert!(txn1.commit().await.is_err()); + } + #[tokio::test] async fn test_txn_si_commit_conflict() { // Setup database with initial data diff --git a/slatedb/src/ops.rs b/slatedb/src/ops.rs index 2e242e09e9..87a31a62b0 100644 --- a/slatedb/src/ops.rs +++ b/slatedb/src/ops.rs @@ -67,6 +67,38 @@ pub trait DbReadOps { options: &ReadOptions, ) -> Result, crate::Error>; + /// Get multiple values from the database with default read options. + /// + /// The returned vector preserves the same order as the input keys, + /// including duplicates. + async fn multi_get(&self, keys: &[K]) -> Result>, crate::Error> + where + K: AsRef<[u8]> + Send + Sync, + { + self.multi_get_with_options(keys, &ReadOptions::default()) + .await + } + + /// Get multiple values from the database with custom read options. + /// + /// The default implementation is correctness-oriented and delegates to + /// repeated point reads. Concrete database handles override this with a + /// batched implementation. + async fn multi_get_with_options( + &self, + keys: &[K], + options: &ReadOptions, + ) -> Result>, crate::Error> + where + K: AsRef<[u8]> + Send + Sync, + { + let mut values = Vec::with_capacity(keys.len()); + for key in keys { + values.push(self.get_with_options(key, options).await?); + } + Ok(values) + } + /// Get a key-value pair from the database with default read options. /// /// Returns the key along with its value and metadata (sequence number, @@ -454,6 +486,22 @@ pub trait DbTransactionOps: DbReadOps { K: AsRef<[u8]>, V: AsRef<[u8]>; + /// Put an owned key-value pair into the transaction with default + /// `PutOptions`, avoiding the copies that [`Self::put`] performs when the + /// caller already has owned [`Bytes`]. + fn put_bytes(&self, key: Bytes, value: Bytes) -> Result<(), crate::Error> { + self.put_bytes_with_options(key, value, &PutOptions::default()) + } + + /// Put an owned key-value pair into the transaction with custom + /// `PutOptions`. + fn put_bytes_with_options( + &self, + key: Bytes, + value: Bytes, + options: &PutOptions, + ) -> Result<(), crate::Error>; + /// Delete a key from the transaction. The delete is buffered in the /// transaction's write batch until commit. fn delete>(&self, key: K) -> Result<(), crate::Error>; diff --git a/slatedb/src/reader.rs b/slatedb/src/reader.rs index c66eb996ff..777a68d00d 100644 --- a/slatedb/src/reader.rs +++ b/slatedb/src/reader.rs @@ -1,30 +1,34 @@ use crate::batch::WriteBatchIterator; +use crate::block_iterator::DataBlockIterator; use crate::bytes_range::BytesRange; use crate::clock::MonotonicClock; use crate::config::{DurabilityLevel, ReadOptions, ScanOptions}; use crate::db_iter::{apply_filters, DbRecencyIterator}; use crate::db_stats::DbStats; +use crate::filter_policy::FilterQuery; +use crate::format::block::Block; use crate::iter::RowEntryIterator; use crate::manifest::{ManifestCore, Segment}; use crate::mem_table::{ImmutableMemtable, KVTable}; use crate::merge_operator::{instrument_merge_operator, MergeOperatorType}; use crate::oracle::Oracle; +use crate::partitioned_keyspace; use crate::segment_iterator::{build_segment_iter, SegmentScanContext}; use crate::sorted_run_iterator::SortedRunIterator; use crate::sst_iter::{SstIterator, SstIteratorOptions}; use crate::tablestore::TableStore; -use crate::types::KeyValue; +use crate::types::{KeyValue, RowEntry, ValueDeletable}; use crate::{error::SlateDBError, DbIterator}; use bytes::Bytes; -use std::collections::VecDeque; +use std::collections::{BTreeMap, HashMap, HashSet, VecDeque}; use std::sync::Arc; pub(crate) trait DbStateReader { fn memtable(&self) -> Arc; /// Returns immutable memtables newest-first. The iterator form permits /// read-only replicas to use a structurally shared persistent chain. - fn imm_memtables(&self) -> Box> + '_>; + fn imm_memtables(&self) -> Box> + Send + '_>; fn core(&self) -> &ManifestCore; } @@ -261,6 +265,549 @@ impl Reader { .transpose() } + pub(crate) async fn multi_get_key_value_with_options( + &self, + keys: &[Bytes], + options: &ReadOptions, + db_state: &(dyn DbStateReader + Sync + Send), + max_seq: Option, + ) -> Result>, SlateDBError> { + self.db_stats.get_requests.increment(1); + if keys.is_empty() { + return Ok(Vec::new()); + } + + let prepared_max_seq = + self.prepare_max_seq(max_seq, options.durability_filter, options.dirty); + let merge_operator_enabled = self.read_merge_operator.is_some(); + + let mut key_to_idx = HashMap::::with_capacity(keys.len()); + let mut unique_keys = Vec::::with_capacity(keys.len()); + let mut output_positions = Vec::>::with_capacity(keys.len()); + + for (output_idx, key) in keys.iter().enumerate() { + if let Some(existing_idx) = key_to_idx.get(key).copied() { + output_positions[existing_idx].push(output_idx); + continue; + } + + let key_idx = unique_keys.len(); + key_to_idx.insert(key.clone(), key_idx); + unique_keys.push(key.clone()); + output_positions.push(vec![output_idx]); + } + + let mut resolved = vec![false; unique_keys.len()]; + let mut values = vec![None; unique_keys.len()]; + let mut fallback_to_point_get = vec![false; unique_keys.len()]; + + self.resolve_rows_from_memtable_for_keys( + db_state.memtable(), + &unique_keys, + prepared_max_seq, + merge_operator_enabled, + &mut resolved, + &mut values, + &mut fallback_to_point_get, + ) + .await?; + + for imm in db_state.imm_memtables() { + if Self::all_done(&resolved, &fallback_to_point_get) { + break; + } + + self.resolve_rows_from_memtable_for_keys( + imm.table(), + &unique_keys, + prepared_max_seq, + merge_operator_enabled, + &mut resolved, + &mut values, + &mut fallback_to_point_get, + ) + .await?; + } + + self.resolve_rows_from_segments_for_keys( + db_state.core(), + &unique_keys, + prepared_max_seq, + options, + merge_operator_enabled, + &mut resolved, + &mut values, + &mut fallback_to_point_get, + ) + .await?; + + for key_idx in fallback_to_point_get + .iter() + .enumerate() + .filter_map(|(idx, should_fallback)| should_fallback.then_some(idx)) + { + values[key_idx] = self + .get_key_value_with_options( + unique_keys[key_idx].as_ref(), + options, + db_state, + None, + max_seq, + ) + .await?; + resolved[key_idx] = true; + } + + let mut result = vec![None; keys.len()]; + for (key_idx, positions) in output_positions.into_iter().enumerate() { + let value = if resolved[key_idx] { + values[key_idx].clone() + } else { + None + }; + + for position in positions { + result[position] = value.clone(); + } + } + + Ok(result) + } + + fn all_done(resolved: &[bool], fallback_to_point_get: &[bool]) -> bool { + resolved + .iter() + .zip(fallback_to_point_get) + .all(|(resolved, fallback)| *resolved || *fallback) + } + + async fn resolve_rows_from_memtable_for_keys( + &self, + table: Arc, + unique_keys: &[Bytes], + max_seq: Option, + merge_operator_enabled: bool, + resolved: &mut [bool], + values: &mut [Option], + fallback_to_point_get: &mut [bool], + ) -> Result<(), SlateDBError> { + if table.is_empty() { + return Ok(()); + } + + for (key_idx, key) in unique_keys.iter().enumerate() { + if resolved[key_idx] || fallback_to_point_get[key_idx] { + continue; + } + + let mut iter = table.range_ascending(key.clone()..=key.clone()); + while let Some(row) = iter.next().await? { + if !Self::row_visible(&row, max_seq) { + continue; + } + + Self::apply_source_row_result( + key_idx, + row, + merge_operator_enabled, + resolved, + values, + fallback_to_point_get, + )?; + break; + } + } + + Ok(()) + } + + async fn resolve_rows_from_segments_for_keys( + &self, + core: &ManifestCore, + unique_keys: &[Bytes], + max_seq: Option, + options: &ReadOptions, + merge_operator_enabled: bool, + resolved: &mut [bool], + values: &mut [Option], + fallback_to_point_get: &mut [bool], + ) -> Result<(), SlateDBError> { + if Self::all_done(resolved, fallback_to_point_get) { + return Ok(()); + } + + let mut groups = BTreeMap::)>::new(); + for (key_idx, key) in unique_keys.iter().enumerate() { + if resolved[key_idx] || fallback_to_point_get[key_idx] { + continue; + } + + let range = BytesRange::from_slice(key.as_ref()..=key.as_ref()); + let segment = match core.select_segments(&range) { + None => core.default_segment(), + Some(segments) => match segments.first() { + Some(segment) => segment.clone(), + None => continue, + }, + }; + + groups + .entry(segment.prefix.clone()) + .or_insert_with(|| (segment, Vec::new())) + .1 + .push(key_idx); + } + + for (_, (segment, key_indices)) in groups { + self.resolve_rows_from_lsm_tree_for_keys( + &segment, + &key_indices, + unique_keys, + max_seq, + options, + merge_operator_enabled, + resolved, + values, + fallback_to_point_get, + ) + .await?; + } + + Ok(()) + } + + async fn resolve_rows_from_lsm_tree_for_keys( + &self, + segment: &Segment, + key_indices: &[usize], + unique_keys: &[Bytes], + max_seq: Option, + options: &ReadOptions, + merge_operator_enabled: bool, + resolved: &mut [bool], + values: &mut [Option], + fallback_to_point_get: &mut [bool], + ) -> Result<(), SlateDBError> { + let mut unresolved = key_indices + .iter() + .copied() + .filter(|key_idx| !resolved[*key_idx] && !fallback_to_point_get[*key_idx]) + .collect::>(); + + for sst in segment.tree.l0.iter() { + if unresolved.is_empty() { + return Ok(()); + } + + self.resolve_rows_from_sst_for_keys( + sst, + &unresolved, + unique_keys, + max_seq, + options, + merge_operator_enabled, + resolved, + values, + fallback_to_point_get, + ) + .await?; + unresolved.retain(|idx| !resolved[*idx] && !fallback_to_point_get[*idx]); + } + + for sr in segment.tree.compacted.iter() { + if unresolved.is_empty() { + return Ok(()); + } + + self.resolve_rows_from_sorted_run_for_keys( + sr, + &unresolved, + unique_keys, + max_seq, + options, + merge_operator_enabled, + resolved, + values, + fallback_to_point_get, + ) + .await?; + unresolved.retain(|idx| !resolved[*idx] && !fallback_to_point_get[*idx]); + } + + Ok(()) + } + + async fn resolve_rows_from_sorted_run_for_keys( + &self, + sorted_run: &crate::db_state::SortedRun, + key_indices: &[usize], + unique_keys: &[Bytes], + max_seq: Option, + options: &ReadOptions, + merge_operator_enabled: bool, + resolved: &mut [bool], + values: &mut [Option], + fallback_to_point_get: &mut [bool], + ) -> Result<(), SlateDBError> { + let mut groups = BTreeMap::)>::new(); + for key_idx in key_indices.iter().copied() { + if resolved[key_idx] || fallback_to_point_get[key_idx] { + continue; + } + + for sst in sorted_run.tables_covering_point_key(unique_keys[key_idx].as_ref()) { + groups + .entry(sst.id) + .or_insert_with(|| (sst.clone(), Vec::new())) + .1 + .push(key_idx); + } + } + + for (_, (sst, group_key_indices)) in groups { + self.resolve_rows_from_sst_for_keys( + &sst, + &group_key_indices, + unique_keys, + max_seq, + options, + merge_operator_enabled, + resolved, + values, + fallback_to_point_get, + ) + .await?; + } + + Ok(()) + } + + async fn resolve_rows_from_sst_for_keys( + &self, + sst: &crate::db_state::SsTableView, + key_indices: &[usize], + unique_keys: &[Bytes], + max_seq: Option, + options: &ReadOptions, + merge_operator_enabled: bool, + resolved: &mut [bool], + values: &mut [Option], + fallback_to_point_get: &mut [bool], + ) -> Result<(), SlateDBError> { + let mut candidate_key_indices = Vec::with_capacity(key_indices.len()); + let filters = self.table_store.read_filters(&sst.sst, true).await?; + + for &key_idx in key_indices { + if resolved[key_idx] || fallback_to_point_get[key_idx] { + continue; + } + + let key = unique_keys[key_idx].as_ref(); + if sst + .calculate_view_range(BytesRange::from_slice(key..=key)) + .is_none() + { + continue; + } + + if filters.is_empty() { + candidate_key_indices.push(key_idx); + continue; + } + + let query = FilterQuery::point(unique_keys[key_idx].clone()) + .with_context(options.filter_context.clone()); + if filters.iter().all(|named| named.filter.might_match(&query)) { + self.db_stats.sst_filter_point_positives.increment(1); + candidate_key_indices.push(key_idx); + } else { + self.db_stats.sst_filter_point_negatives.increment(1); + } + } + + if candidate_key_indices.is_empty() { + return Ok(()); + } + + let index = self.table_store.read_index(&sst.sst, true).await?; + let block_to_key_indices = { + let index_ref = index.borrow(); + let mut block_to_key_indices = BTreeMap::>::new(); + for &key_idx in &candidate_key_indices { + let key = unique_keys[key_idx].as_ref(); + let start_block = + partitioned_keyspace::first_partition_including_or_after_key(&index_ref, key); + let end_block_exclusive = + partitioned_keyspace::last_partition_including_key(&index_ref, key) + .map(|last| last + 1) + .unwrap_or(start_block); + + if start_block >= end_block_exclusive { + continue; + } + + for block_idx in start_block..end_block_exclusive { + block_to_key_indices + .entry(block_idx) + .or_default() + .push(key_idx); + } + } + block_to_key_indices + }; + + if block_to_key_indices.is_empty() { + return Ok(()); + } + + let mut block_ranges = Vec::new(); + let mut range_start = None; + let mut previous_block = 0; + for &block_idx in block_to_key_indices.keys() { + if let Some(start) = range_start { + if block_idx == previous_block + 1 { + previous_block = block_idx; + } else { + block_ranges.push(start..(previous_block + 1)); + range_start = Some(block_idx); + previous_block = block_idx; + } + } else { + range_start = Some(block_idx); + previous_block = block_idx; + } + } + if let Some(start) = range_start { + block_ranges.push(start..(previous_block + 1)); + } + + let mut found_in_sst = HashSet::::new(); + for block_range in block_ranges { + let blocks = self + .table_store + .read_blocks_using_index( + &sst.sst, + index.clone(), + block_range.clone(), + options.cache_blocks, + ) + .await?; + + for (offset, block) in blocks.into_iter().enumerate() { + let block_idx = block_range.start + offset; + let Some(key_idxs) = block_to_key_indices.get(&block_idx) else { + continue; + }; + + self.resolve_rows_from_block_for_keys( + block, + sst.sst.format_version, + key_idxs, + unique_keys, + max_seq, + merge_operator_enabled, + resolved, + values, + fallback_to_point_get, + &mut found_in_sst, + ) + .await?; + } + } + + if !filters.is_empty() { + let false_positives = candidate_key_indices + .iter() + .filter(|key_idx| !found_in_sst.contains(key_idx)) + .count() as u64; + if false_positives > 0 { + self.db_stats + .sst_filter_point_false_positives + .increment(false_positives); + } + } + + Ok(()) + } + + async fn resolve_rows_from_block_for_keys( + &self, + block: Arc, + sst_version: u16, + key_indices: &[usize], + unique_keys: &[Bytes], + max_seq: Option, + merge_operator_enabled: bool, + resolved: &mut [bool], + values: &mut [Option], + fallback_to_point_get: &mut [bool], + found_in_sst: &mut HashSet, + ) -> Result<(), SlateDBError> { + let key_lookup = key_indices + .iter() + .copied() + .map(|key_idx| (unique_keys[key_idx].clone(), key_idx)) + .collect::>(); + + let mut iter = + DataBlockIterator::new(block, sst_version, crate::iter::IterationOrder::Ascending)?; + while let Some(row) = iter.next().await? { + let Some(&key_idx) = key_lookup.get(&row.key) else { + continue; + }; + if found_in_sst.contains(&key_idx) || !Self::row_visible(&row, max_seq) { + continue; + } + + found_in_sst.insert(key_idx); + Self::apply_source_row_result( + key_idx, + row, + merge_operator_enabled, + resolved, + values, + fallback_to_point_get, + )?; + } + + Ok(()) + } + + fn row_visible(row: &RowEntry, max_seq: Option) -> bool { + match max_seq { + Some(max_seq) => row.seq <= max_seq, + None => true, + } + } + + fn apply_source_row_result( + key_idx: usize, + row: RowEntry, + merge_operator_enabled: bool, + resolved: &mut [bool], + values: &mut [Option], + fallback_to_point_get: &mut [bool], + ) -> Result<(), SlateDBError> { + match &row.value { + ValueDeletable::Value(_) => { + resolved[key_idx] = true; + values[key_idx] = Some(KeyValue::from(row)); + } + ValueDeletable::Tombstone => { + resolved[key_idx] = true; + values[key_idx] = None; + } + ValueDeletable::Merge(_) => { + if !merge_operator_enabled { + return Err(SlateDBError::MergeOperatorMissing); + } + fallback_to_point_get[key_idx] = true; + } + } + + Ok(()) + } + /// Create an iterator over a key range. /// /// Produces a merged iterator over the provided `write_batch` (if any), @@ -607,7 +1154,9 @@ mod tests { self.memtable.clone() } - fn imm_memtables(&self) -> Box> + '_> { + fn imm_memtables( + &self, + ) -> Box> + Send + '_> { Box::new(self.imm_memtable.iter().cloned()) } From 53ab758f3f78ca1d07a486d1b30d743cd3cc42cb Mon Sep 17 00:00:00 2001 From: Matthew Sanetra <41018997+matthewsanetra@users.noreply.github.com> Date: Thu, 16 Jul 2026 21:54:31 +0100 Subject: [PATCH 54/65] ci: drop Windows test runner --- .github/workflows/pr.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/workflows/pr.yaml b/.github/workflows/pr.yaml index 90f1a5f355..7c09615581 100644 --- a/.github/workflows/pr.yaml +++ b/.github/workflows/pr.yaml @@ -121,7 +121,7 @@ jobs: strategy: fail-fast: false matrix: - os: [macos-latest, windows-latest] + os: [macos-latest] runs-on: ${{ matrix.os }} steps: - uses: actions/checkout@v4 From 539f1e8eabf4b9dbb201b45c49d2a6526beccea0 Mon Sep 17 00:00:00 2001 From: xav-db Date: Wed, 29 Jul 2026 13:25:17 +0100 Subject: [PATCH 55/65] Add SlateDB cache usage snapshots --- Cargo.toml | 1 + slatedb/Cargo.toml | 3 +- .../src/cached_object_store/object_store.rs | 5 + slatedb/src/cached_object_store/storage.rs | 6 + slatedb/src/cached_object_store/storage_fs.rs | 281 ++++++++++++++--- slatedb/src/db.rs | 25 ++ slatedb/src/db/builder.rs | 8 +- slatedb/src/db_cache/foyer.rs | 15 +- slatedb/src/db_cache/foyer_hybrid.rs | 287 +++++++++++++++++- slatedb/src/db_cache/mod.rs | 170 ++++++++++- slatedb/src/db_reader.rs | 26 ++ 11 files changed, 772 insertions(+), 55 deletions(-) diff --git a/Cargo.toml b/Cargo.toml index 6d0a7498c2..c16cae2117 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -57,6 +57,7 @@ log = "0.4.27" lru = "0.18" lz4_flex = "0.11.5" moka = "0.12.8" +mixtrics = "0.2" object_store = "0.14.0" ouroboros = "0.18" parking_lot = "0.12.4" diff --git a/slatedb/Cargo.toml b/slatedb/Cargo.toml index 64e0f81272..854ad9e657 100644 --- a/slatedb/Cargo.toml +++ b/slatedb/Cargo.toml @@ -34,6 +34,7 @@ log = { workspace = true } lru = { workspace = true } lz4_flex = { workspace = true, optional = true } moka = { workspace = true, features = ["future"], optional = true } +mixtrics = { workspace = true, optional = true } object_store = { workspace = true } ouroboros = { workspace = true } parking_lot = { workspace = true } @@ -101,7 +102,7 @@ lz4 = ["dep:lz4_flex"] zstd = ["dep:zstd"] wal_disable = [] moka = ["dep:moka"] -foyer = ["dep:foyer"] +foyer = ["dep:foyer", "dep:mixtrics"] bench-internal = [] test-util = [ "tokio/test-util", diff --git a/slatedb/src/cached_object_store/object_store.rs b/slatedb/src/cached_object_store/object_store.rs index 1cf2fbdabb..47dbaf1f3f 100644 --- a/slatedb/src/cached_object_store/object_store.rs +++ b/slatedb/src/cached_object_store/object_store.rs @@ -23,6 +23,7 @@ use std::{ops::Range, sync::Arc}; use crate::single_flight::SingleFlight; use crate::cached_object_store::storage::{LocalCacheStorage, PartID}; +use crate::db_cache::CacheUsageSnapshot; use crate::error::SlateDBError; use crate::utils::build_concurrent; use log::warn; @@ -73,6 +74,10 @@ pub struct CachedObjectStore { } impl CachedObjectStore { + pub(crate) fn usage_snapshot(&self) -> CacheUsageSnapshot { + self.cache_storage.usage_snapshot() + } + pub(crate) fn new( object_store: Arc, cache_storage: Arc, diff --git a/slatedb/src/cached_object_store/storage.rs b/slatedb/src/cached_object_store/storage.rs index 5ca2af16c2..a487bd4362 100644 --- a/slatedb/src/cached_object_store/storage.rs +++ b/slatedb/src/cached_object_store/storage.rs @@ -4,6 +4,8 @@ use object_store::{path::Path, Attribute, Attributes, ObjectMeta}; use serde::{Deserialize, Serialize}; use std::{collections::HashMap, fmt::Display, ops::Range}; +use crate::db_cache::CacheUsageSnapshot; + #[derive(Debug, Clone, Serialize, Deserialize)] pub struct LocalCacheHead { pub location: String, @@ -74,6 +76,10 @@ pub trait LocalCacheStorage: Send + Sync + std::fmt::Debug + Display + 'static { fn entry(&self, location: &Path, part_size: usize) -> Box; async fn start_evictor(&self); + + fn usage_snapshot(&self) -> CacheUsageSnapshot { + CacheUsageSnapshot::Unavailable + } } #[async_trait] diff --git a/slatedb/src/cached_object_store/storage_fs.rs b/slatedb/src/cached_object_store/storage_fs.rs index 8ace3f99d1..ea1eb383b1 100644 --- a/slatedb/src/cached_object_store/storage_fs.rs +++ b/slatedb/src/cached_object_store/storage_fs.rs @@ -16,10 +16,11 @@ use std::ops::Range; use std::sync::atomic::{AtomicBool, AtomicI64, AtomicU64, Ordering}; use std::sync::Arc; use std::time::Duration; -use tokio::sync::{Mutex, OnceCell}; +use tokio::sync::{Mutex, Notify, OnceCell}; use walkdir::WalkDir; use crate::cached_object_store::storage::{LocalCacheEntry, LocalCacheHead, LocalCacheStorage}; +use crate::db_cache::CacheUsageSnapshot; use crate::utils::format_bytes_si; /// A cached file handle node. Callers that obtain an `Arc` @@ -150,7 +151,7 @@ fn read_exact_at_offset(file: &std::fs::File, buf: &mut [u8], offset: u64) -> st #[derive(Debug)] pub struct FsCacheStorage { root_folder: std::path::PathBuf, - evictor: Option>, + evictor: Arc, rand: Arc, file_handle_cache: FileHandleCache, } @@ -166,17 +167,15 @@ impl FsCacheStorage { max_open_file_handles: usize, ) -> Self { let file_handle_cache = FileHandleCache::new(max_open_file_handles); - let evictor = max_cache_size_bytes.map(|max_cache_size_bytes| { - Arc::new(FsCacheEvictor::new( - root_folder.clone(), - max_cache_size_bytes, - scan_interval, - stats, - system_clock, - rand.clone(), - file_handle_cache.clone(), - )) - }); + let evictor = Arc::new(FsCacheEvictor::new( + root_folder.clone(), + max_cache_size_bytes, + scan_interval, + stats, + system_clock, + rand.clone(), + file_handle_cache.clone(), + )); Self { root_folder, @@ -198,7 +197,7 @@ impl LocalCacheStorage for FsCacheStorage { Box::new(FsCacheEntry { root_folder: self.root_folder.clone(), location: location.clone(), - evictor: self.evictor.clone(), + evictor: Some(self.evictor.clone()), part_size, rand: self.rand.clone(), file_handle_cache: self.file_handle_cache.clone(), @@ -206,9 +205,11 @@ impl LocalCacheStorage for FsCacheStorage { } async fn start_evictor(&self) { - if let Some(evictor) = &self.evictor { - evictor.start().await - } + self.evictor.start().await + } + + fn usage_snapshot(&self) -> CacheUsageSnapshot { + self.evictor.usage_snapshot() } } @@ -549,7 +550,7 @@ const QUEUE_FULL_LOG_INTERVAL_MS: i64 = 30_000; #[derive(Debug)] struct FsCacheEvictor { root_folder: std::path::PathBuf, - max_cache_size_bytes: usize, + max_cache_size_bytes: Option, scan_interval: Option, tx: tokio::sync::mpsc::Sender, rx: Mutex>>, @@ -562,12 +563,35 @@ struct FsCacheEvictor { system_clock: Arc, rand: Arc, file_handle_cache: FileHandleCache, + usage: Arc, + reconcile_notify: Arc, +} + +#[derive(Debug)] +struct FsCacheUsage { + initialized: AtomicBool, + used_bytes: AtomicU64, + capacity_bytes: Option, +} + +impl FsCacheUsage { + fn snapshot(&self) -> CacheUsageSnapshot { + if !self.initialized.load(Ordering::Acquire) { + return CacheUsageSnapshot::Initializing { + capacity_bytes: self.capacity_bytes, + }; + } + CacheUsageSnapshot::Ready { + used_bytes: self.used_bytes.load(Ordering::Acquire), + capacity_bytes: self.capacity_bytes, + } + } } impl FsCacheEvictor { fn new( root_folder: std::path::PathBuf, - max_cache_size_bytes: usize, + max_cache_size_bytes: Option, scan_interval: Option, stats: Arc, system_clock: Arc, @@ -590,16 +614,23 @@ impl FsCacheEvictor { system_clock, rand, file_handle_cache, + usage: Arc::new(FsCacheUsage { + initialized: AtomicBool::new(false), + used_bytes: AtomicU64::new(0), + capacity_bytes: max_cache_size_bytes.map(|bytes| bytes as u64), + }), + reconcile_notify: Arc::new(Notify::new()), } } async fn start(&self) { - let inner = Arc::new(FsCacheEvictorInner::new( + let inner = Arc::new(FsCacheEvictorInner::new_with_usage( self.root_folder.clone(), self.max_cache_size_bytes, self.stats.clone(), self.rand.clone(), self.file_handle_cache.clone(), + self.usage.clone(), )); let guard = self.rx.lock(); @@ -614,6 +645,7 @@ impl FsCacheEvictor { inner.clone(), self.scan_interval, self.system_clock.clone(), + self.reconcile_notify.clone(), ))) .ok(); @@ -631,6 +663,10 @@ impl FsCacheEvictor { self.started.load(Ordering::Acquire) } + fn usage_snapshot(&self) -> CacheUsageSnapshot { + self.usage.snapshot() + } + async fn background_evict( inner: Arc, mut rx: tokio::sync::mpsc::Receiver, @@ -662,14 +698,21 @@ impl FsCacheEvictor { inner: Arc, scan_interval: Option, system_clock: Arc, + reconcile_notify: Arc, ) { inner.clone().scan_entries(true).await; - if let Some(scan_interval) = scan_interval { - loop { - system_clock.clone().sleep(scan_interval).await; - inner.clone().scan_entries(true).await; + loop { + if let Some(scan_interval) = scan_interval { + let clock = system_clock.clone(); + tokio::select! { + () = clock.sleep(scan_interval) => {} + () = reconcile_notify.notified() => {} + } + } else { + reconcile_notify.notified().await; } + inner.clone().scan_entries(true).await; } } @@ -685,6 +728,8 @@ impl FsCacheEvictor { match self.tx.try_send((path, access)) { Ok(()) => true, Err(tokio::sync::mpsc::error::TrySendError::Full(_)) => { + self.usage.initialized.store(false, Ordering::Release); + self.reconcile_notify.notify_one(); self.queue_full_count.fetch_add(1, Ordering::AcqRel); let now_ms = self.system_clock.now().timestamp_millis(); let last_log_ms = self.last_queue_full_log_ms.load(Ordering::Acquire); @@ -702,7 +747,11 @@ impl FsCacheEvictor { } false } - Err(tokio::sync::mpsc::error::TrySendError::Closed(_)) => false, + Err(tokio::sync::mpsc::error::TrySendError::Closed(_)) => { + self.usage.initialized.store(false, Ordering::Release); + self.reconcile_notify.notify_one(); + false + } } } } @@ -746,29 +795,52 @@ impl CacheState { #[derive(Debug)] struct FsCacheEvictorInner { root_folder: std::path::PathBuf, - max_cache_size_bytes: usize, + max_cache_size_bytes: Option, track_lock: Mutex<()>, cache_state: Mutex, - cache_size_bytes: AtomicU64, + usage: Arc, stats: Arc, rand: Arc, file_handle_cache: FileHandleCache, } impl FsCacheEvictorInner { + #[cfg(test)] fn new( root_folder: std::path::PathBuf, max_cache_size_bytes: usize, stats: Arc, rand: Arc, file_handle_cache: FileHandleCache, + ) -> Self { + Self::new_with_usage( + root_folder, + Some(max_cache_size_bytes), + stats, + rand, + file_handle_cache, + Arc::new(FsCacheUsage { + initialized: AtomicBool::new(false), + used_bytes: AtomicU64::new(0), + capacity_bytes: Some(max_cache_size_bytes as u64), + }), + ) + } + + fn new_with_usage( + root_folder: std::path::PathBuf, + max_cache_size_bytes: Option, + stats: Arc, + rand: Arc, + file_handle_cache: FileHandleCache, + usage: Arc, ) -> Self { Self { root_folder, max_cache_size_bytes, track_lock: Mutex::new(()), cache_state: Mutex::new(CacheState::default()), - cache_size_bytes: AtomicU64::new(0_u64), + usage, stats, rand, file_handle_cache, @@ -793,6 +865,7 @@ impl FsCacheEvictorInner { .await .unwrap_or_default(); + let scanned_paths: HashSet<_> = paths.iter().cloned().collect(); for path in paths { let metadata = match tokio::fs::metadata(&path).await { Ok(metadata) => metadata, @@ -816,6 +889,34 @@ impl FsCacheEvictorInner { self.track_entry_accessed(path, bytes, atime, evict).await; } + + let tracked_paths = self.cache_state.lock().await.keys.clone(); + let mut missing_paths = Vec::new(); + for path in tracked_paths { + if !scanned_paths.contains(&path) + && !tokio::fs::try_exists(&path).await.unwrap_or(false) + { + missing_paths.push(path); + } + } + if !missing_paths.is_empty() { + let _track_guard = self.track_lock.lock().await; + let mut cache_state = self.cache_state.lock().await; + for path in missing_paths { + if let Some(removed) = cache_state.remove_entry(&path) { + self.usage + .used_bytes + .fetch_sub(removed.size_bytes as u64, Ordering::AcqRel); + } + } + self.stats + .object_store_cache_keys + .set(cache_state.entries.len() as i64); + self.stats + .object_store_cache_bytes + .set(self.usage.used_bytes.load(Ordering::Relaxed) as i64); + } + self.usage.initialized.store(true, Ordering::Release); } /// track the cache entry access, and evict the cache files when the cache size exceeds the limit if evict is true, @@ -836,6 +937,18 @@ impl FsCacheEvictorInner { match cache_state.entries.get_mut(&path) { Some(entry) => { entry.access_time = accessed_time; + if entry.size_bytes != bytes { + if bytes > entry.size_bytes { + self.usage + .used_bytes + .fetch_add((bytes - entry.size_bytes) as u64, Ordering::AcqRel); + } else { + self.usage + .used_bytes + .fetch_sub((entry.size_bytes - bytes) as u64, Ordering::AcqRel); + } + entry.size_bytes = bytes; + } } None => { let key_index = cache_state.keys.len(); @@ -848,7 +961,8 @@ impl FsCacheEvictorInner { key_index, }, ); - self.cache_size_bytes + self.usage + .used_bytes .fetch_add(bytes as u64, Ordering::SeqCst); } } @@ -858,10 +972,13 @@ impl FsCacheEvictorInner { self.stats.object_store_cache_keys.set(entry_count as i64); self.stats .object_store_cache_bytes - .set(self.cache_size_bytes.load(Ordering::Relaxed) as i64); + .set(self.usage.used_bytes.load(Ordering::Relaxed) as i64); + let Some(max_cache_size_bytes) = self.max_cache_size_bytes else { + return 0; + }; // if the cache size is still below the limit, do nothing - if self.cache_size_bytes.load(Ordering::Relaxed) <= self.max_cache_size_bytes as u64 { + if self.usage.used_bytes.load(Ordering::Relaxed) <= max_cache_size_bytes as u64 { return 0; } // TODO: check the disk space ratio here, if the disk space is low, also triggers evict. @@ -873,10 +990,10 @@ impl FsCacheEvictorInner { // It's ok to call evict after inserting the new entry, because we will evict entries with eailer `accessed_time`. // This ensures that the newly added entry will not be evicted immediately. let evicted_bytes: usize = if evict - && self.cache_size_bytes.load(Ordering::Relaxed) > self.max_cache_size_bytes as u64 + && self.usage.used_bytes.load(Ordering::Relaxed) > max_cache_size_bytes as u64 { // We sacrifice floating-point precision error to prevent possible overflow(i.e. self.max_cache_size_bytes * 9 / 10). - let target_size = ((self.max_cache_size_bytes as f64) * 0.9) as u64; + let target_size = ((max_cache_size_bytes as f64) * 0.9) as u64; self.evict_to_target_size(target_size).await } else { 0 @@ -892,11 +1009,11 @@ impl FsCacheEvictorInner { let picked_targets = self.pick_evict_targets(target_size).await; if picked_targets.is_empty() { - if self.cache_size_bytes.load(Ordering::Relaxed) > target_size { + if self.usage.used_bytes.load(Ordering::Relaxed) > target_size { warn!( "cache_size_bytes still exceeds max_cache_size_bytes but no more entries can be evicted(cache_size_bytes={}, max_cache_size_bytes={})", - format_bytes_si(self.cache_size_bytes.load(Ordering::Relaxed)), - format_bytes_si(self.max_cache_size_bytes as u64) + format_bytes_si(self.usage.used_bytes.load(Ordering::Relaxed)), + format_bytes_si(self.max_cache_size_bytes.unwrap_or_default() as u64) ); } return 0; @@ -940,7 +1057,8 @@ impl FsCacheEvictorInner { for (target, target_bytes) in deleted_targets.iter() { if cache_state.remove_entry(target).is_some() { - self.cache_size_bytes + self.usage + .used_bytes .fetch_sub(*target_bytes as u64, Ordering::SeqCst); total_bytes += target_bytes; } @@ -959,7 +1077,7 @@ impl FsCacheEvictorInner { self.stats.object_store_cache_keys.set(entry_count as i64); self.stats .object_store_cache_bytes - .set(self.cache_size_bytes.load(Ordering::Relaxed) as i64); + .set(self.usage.used_bytes.load(Ordering::Relaxed) as i64); total_evicted_bytes } @@ -977,7 +1095,8 @@ impl FsCacheEvictorInner { let mut cache_state = self.cache_state.lock().await; for deleted_entry in deleted_entries { if let Some(removed) = cache_state.remove_entry(&deleted_entry) { - self.cache_size_bytes + self.usage + .used_bytes .fetch_sub(removed.size_bytes as u64, Ordering::SeqCst); } } @@ -988,7 +1107,7 @@ impl FsCacheEvictorInner { self.stats.object_store_cache_keys.set(entry_count as i64); self.stats .object_store_cache_bytes - .set(self.cache_size_bytes.load(Ordering::Relaxed) as i64); + .set(self.usage.used_bytes.load(Ordering::Relaxed) as i64); } /// Pick multiple eviction targets in a single pass using pick-of-2 strategy, which is an approximation @@ -1005,7 +1124,7 @@ impl FsCacheEvictorInner { let mut targets = Vec::new(); // Track the simulated cache size during eviction but do not modify the actual cache size until // after files are deleted. - let mut simulated_size = self.cache_size_bytes.load(Ordering::Relaxed); + let mut simulated_size = self.usage.used_bytes.load(Ordering::Relaxed); // Track which indices have been selected for eviction let mut picked_indices: HashSet = HashSet::new(); @@ -1238,7 +1357,7 @@ mod tests { let evictor = FsCacheEvictor::new( temp_dir.path().to_path_buf(), - 1024, + Some(1024), None, Arc::new(CachedObjectStoreStats::new(&recorder)), Arc::new(DefaultSystemClock::new()), @@ -1317,13 +1436,87 @@ mod tests { // rescan two times, the cache size should be 2049 unchanged evictor.clone().scan_entries(false).await; - let cache_size_bytes = evictor.cache_size_bytes.load(Ordering::SeqCst); + let cache_size_bytes = evictor.usage.used_bytes.load(Ordering::SeqCst); assert_eq!(cache_size_bytes, 2049); evictor.clone().scan_entries(false).await; - let cache_size_bytes = evictor.cache_size_bytes.load(Ordering::SeqCst); + let cache_size_bytes = evictor.usage.used_bytes.load(Ordering::SeqCst); assert_eq!(cache_size_bytes, 2049); } + #[tokio::test] + async fn usage_snapshot_reconciles_bounded_and_unbounded_cache_files() { + let temp_dir = tempfile::Builder::new() + .prefix("objstore_cache_usage_snapshot_") + .tempdir() + .unwrap(); + let recorder = slatedb_common::metrics::MetricsRecorderHelper::noop(); + let usage = Arc::new(FsCacheUsage { + initialized: AtomicBool::new(false), + used_bytes: AtomicU64::new(0), + capacity_bytes: None, + }); + let evictor = Arc::new(FsCacheEvictorInner::new_with_usage( + temp_dir.path().to_path_buf(), + None, + Arc::new(CachedObjectStoreStats::new(&recorder)), + Arc::new(DbRand::default()), + FileHandleCache::new(1000), + usage.clone(), + )); + assert_eq!( + usage.snapshot(), + CacheUsageSnapshot::Initializing { + capacity_bytes: None, + } + ); + + let path = gen_rand_file(temp_dir.path(), "file", 17); + evictor.clone().scan_entries(false).await; + assert_eq!( + usage.snapshot(), + CacheUsageSnapshot::Ready { + used_bytes: 17, + capacity_bytes: None, + } + ); + + gen_rand_file(temp_dir.path(), "file", 29); + evictor.clone().scan_entries(false).await; + assert_eq!( + usage.snapshot(), + CacheUsageSnapshot::Ready { + used_bytes: 29, + capacity_bytes: None, + } + ); + + std::fs::remove_file(path).unwrap(); + evictor.scan_entries(false).await; + assert_eq!( + usage.snapshot(), + CacheUsageSnapshot::Ready { + used_bytes: 0, + capacity_bytes: None, + } + ); + + let bounded = FsCacheStorage::new( + temp_dir.path().to_path_buf(), + Some(1024), + None, + Arc::new(CachedObjectStoreStats::new(&recorder)), + Arc::new(DefaultSystemClock::new()), + Arc::new(DbRand::default()), + 10, + ); + assert_eq!( + bounded.usage_snapshot(), + CacheUsageSnapshot::Initializing { + capacity_bytes: Some(1024), + } + ); + } + #[rstest::rstest] // Basic case: 2 keys, nothing picked, no exclusion #[case(&[0, 1], &[], None, &[0, 1])] diff --git a/slatedb/src/db.rs b/slatedb/src/db.rs index 4c331139ea..ae32de0331 100644 --- a/slatedb/src/db.rs +++ b/slatedb/src/db.rs @@ -620,9 +620,34 @@ impl DbInner { pub struct Db { pub(crate) inner: Arc, task_executor: Arc, + pub(crate) object_store_cache: Option>, } impl Db { + /// Returns a synchronous point-in-time cache accounting snapshot. + /// + /// This method reads only in-memory counters and never performs object-store + /// or filesystem I/O. + pub fn cache_usage_snapshot(&self) -> crate::db_cache::SlateDbCacheUsageSnapshot { + let db_cache = self.inner.table_store.cache().map_or_else( + || crate::db_cache::DbCacheUsageSnapshot { + memory: crate::db_cache::CacheUsageSnapshot::Disabled, + disk: crate::db_cache::CacheUsageSnapshot::Disabled, + }, + |cache| cache.usage_snapshot(), + ); + let object_store = self + .object_store_cache + .as_ref() + .map_or(crate::db_cache::CacheUsageSnapshot::Disabled, |cache| { + cache.usage_snapshot() + }); + crate::db_cache::SlateDbCacheUsageSnapshot { + db_cache, + object_store, + } + } + /// Open a new database with default options. /// /// ## Arguments diff --git a/slatedb/src/db/builder.rs b/slatedb/src/db/builder.rs index 4eecf2ffc3..fe7d573d30 100644 --- a/slatedb/src/db/builder.rs +++ b/slatedb/src/db/builder.rs @@ -846,9 +846,9 @@ impl> DbBuilder

{ inner.replay_wal(replay_iterator).await?; // Preload cache if enabled - if let Some(cached_obj_store) = cached_object_store { + if let Some(cached_obj_store) = &cached_object_store { inner - .preload_cache(&cached_obj_store, &path_resolver) + .preload_cache(cached_obj_store, &path_resolver) .await?; } @@ -856,6 +856,7 @@ impl> DbBuilder

{ Ok(Db { inner, task_executor, + object_store_cache: cached_object_store, }) } } @@ -1940,7 +1941,7 @@ impl> DbReaderBuilder

{ TableStoreKind::Reader, )); - let reader = DbReader::open_internal( + let mut reader = DbReader::open_internal( manifest_store, table_store, wal_store, @@ -1959,6 +1960,7 @@ impl> DbReaderBuilder

{ if let Some(cached) = &maybe_cached { reader.preload_cache(cached, path).await?; } + reader.object_store_cache = maybe_cached; Ok(reader) } diff --git a/slatedb/src/db_cache/foyer.rs b/slatedb/src/db_cache/foyer.rs index 66d7e54a7f..af83148bfa 100644 --- a/slatedb/src/db_cache/foyer.rs +++ b/slatedb/src/db_cache/foyer.rs @@ -31,7 +31,10 @@ //! ``` //! -use crate::db_cache::{CacheLoader, CachedEntry, CachedKey, DbCache, DEFAULT_MAX_CAPACITY}; +use crate::db_cache::{ + CacheLoader, CacheUsageSnapshot, CachedEntry, CachedKey, DbCache, DbCacheUsageSnapshot, + DEFAULT_MAX_CAPACITY, +}; use crate::error::SlateDBError; use async_trait::async_trait; use std::sync::Arc; @@ -127,6 +130,16 @@ impl DbCache for FoyerCache { 0 } + fn usage_snapshot(&self) -> DbCacheUsageSnapshot { + DbCacheUsageSnapshot { + memory: CacheUsageSnapshot::Ready { + used_bytes: self.inner.usage() as u64, + capacity_bytes: Some(self.inner.capacity() as u64), + }, + disk: CacheUsageSnapshot::Unavailable, + } + } + async fn fetch_block( &self, key: CachedKey, diff --git a/slatedb/src/db_cache/foyer_hybrid.rs b/slatedb/src/db_cache/foyer_hybrid.rs index f40e9d8986..773edf3833 100644 --- a/slatedb/src/db_cache/foyer_hybrid.rs +++ b/slatedb/src/db_cache/foyer_hybrid.rs @@ -71,21 +71,236 @@ //! use crate::{ - db_cache::{CacheLoader, CachedEntry, CachedKey, DbCache}, + db_cache::{ + CacheLoader, CacheUsageSnapshot, CachedEntry, CachedKey, DbCache, DbCacheUsageSnapshot, + }, error::SlateDBError, utils::format_bytes_si, }; use async_trait::async_trait; use log::info; -use std::sync::Arc; +use mixtrics::{ + metrics::{ + BoxedCounterVec, BoxedGauge, BoxedGaugeVec, BoxedHistogramVec, GaugeOps, GaugeVecOps, + RegistryOps, + }, + registry::noop::NoopMetricsRegistry, +}; +use std::{ + borrow::Cow, + sync::{ + atomic::{AtomicU64, Ordering}, + Arc, + }, +}; + +const FOYER_BLOCK_GAUGE: &str = "foyer_storage_block_engine_block"; +const FOYER_BLOCK_SIZE_GAUGE: &str = "foyer_storage_block_engine_block_size_bytes"; + +#[derive(Debug, Default)] +struct FoyerHybridCacheMetricValues { + clean_blocks: AtomicU64, + writing_blocks: AtomicU64, + evictable_blocks: AtomicU64, + reclaiming_blocks: AtomicU64, + block_size_bytes: AtomicU64, +} + +/// Retained Foyer block-engine gauges used for synchronous cache accounting. +/// +/// Install [`Self::registry`] on the same `HybridCacheBuilder` whose cache is +/// passed to [`FoyerHybridCache::new_with_cache_and_metrics`]. +#[derive(Clone, Debug, Default)] +pub struct FoyerHybridCacheMetrics { + values: Arc, +} + +impl FoyerHybridCacheMetrics { + /// Build an empty metrics handle. + pub fn new() -> Self { + Self::default() + } + + /// Build the registry passed to `HybridCacheBuilder::with_metrics_registry`. + pub fn registry(&self) -> mixtrics::metrics::BoxedRegistry { + Box::new(FoyerCacheMetricsRegistry { + values: Arc::clone(&self.values), + }) + } + + fn disk_usage_snapshot(&self) -> CacheUsageSnapshot { + let block_size_bytes = self.values.block_size_bytes.load(Ordering::Relaxed); + if block_size_bytes == 0 { + return CacheUsageSnapshot::Unavailable; + } + let clean_blocks = self.values.clean_blocks.load(Ordering::Relaxed); + let writing_blocks = self.values.writing_blocks.load(Ordering::Relaxed); + let evictable_blocks = self.values.evictable_blocks.load(Ordering::Relaxed); + let reclaiming_blocks = self.values.reclaiming_blocks.load(Ordering::Relaxed); + let used_blocks = writing_blocks + .saturating_add(evictable_blocks) + .saturating_add(reclaiming_blocks); + let total_blocks = clean_blocks.saturating_add(used_blocks); + CacheUsageSnapshot::Ready { + used_bytes: used_blocks.saturating_mul(block_size_bytes), + capacity_bytes: Some(total_blocks.saturating_mul(block_size_bytes)), + } + } +} + +#[derive(Debug)] +struct FoyerCacheMetricsRegistry { + values: Arc, +} + +impl RegistryOps for FoyerCacheMetricsRegistry { + fn register_counter_vec( + &self, + _name: Cow<'static, str>, + _desc: Cow<'static, str>, + _label_names: &'static [&'static str], + ) -> BoxedCounterVec { + Box::new(NoopMetricsRegistry) + } + + fn register_gauge_vec( + &self, + name: Cow<'static, str>, + _desc: Cow<'static, str>, + _label_names: &'static [&'static str], + ) -> BoxedGaugeVec { + let kind = match name.as_ref() { + FOYER_BLOCK_GAUGE => TrackedGaugeKind::BlockState, + FOYER_BLOCK_SIZE_GAUGE => TrackedGaugeKind::BlockSize, + _ => return Box::new(NoopMetricsRegistry), + }; + Box::new(FoyerTrackedGaugeVec { + kind, + values: Arc::clone(&self.values), + }) + } + + fn register_histogram_vec( + &self, + _name: Cow<'static, str>, + _desc: Cow<'static, str>, + _label_names: &'static [&'static str], + ) -> BoxedHistogramVec { + Box::new(NoopMetricsRegistry) + } + + fn register_histogram_vec_with_buckets( + &self, + _name: Cow<'static, str>, + _desc: Cow<'static, str>, + _label_names: &'static [&'static str], + _buckets: Vec, + ) -> BoxedHistogramVec { + Box::new(NoopMetricsRegistry) + } +} + +#[derive(Debug, Clone, Copy)] +enum TrackedGaugeKind { + BlockState, + BlockSize, +} + +#[derive(Debug)] +struct FoyerTrackedGaugeVec { + kind: TrackedGaugeKind, + values: Arc, +} + +impl GaugeVecOps for FoyerTrackedGaugeVec { + fn gauge(&self, labels: &[Cow<'static, str>]) -> BoxedGauge { + Box::new(AtomicGauge { + value: Arc::clone(&self.values), + field: match self.kind { + TrackedGaugeKind::BlockSize => AtomicGaugeField::BlockSize, + TrackedGaugeKind::BlockState => match labels.get(1).map(Cow::as_ref) { + Some("clean") => AtomicGaugeField::Clean, + Some("writing") => AtomicGaugeField::Writing, + Some("evictable") => AtomicGaugeField::Evictable, + Some("reclaiming") => AtomicGaugeField::Reclaiming, + _ => return Box::new(NoopMetricsRegistry), + }, + }, + }) + } +} + +#[derive(Debug, Clone, Copy)] +enum AtomicGaugeField { + Clean, + Writing, + Evictable, + Reclaiming, + BlockSize, +} + +#[derive(Debug)] +struct AtomicGauge { + value: Arc, + field: AtomicGaugeField, +} + +impl AtomicGauge { + fn atomic(&self) -> &AtomicU64 { + match self.field { + AtomicGaugeField::Clean => &self.value.clean_blocks, + AtomicGaugeField::Writing => &self.value.writing_blocks, + AtomicGaugeField::Evictable => &self.value.evictable_blocks, + AtomicGaugeField::Reclaiming => &self.value.reclaiming_blocks, + AtomicGaugeField::BlockSize => &self.value.block_size_bytes, + } + } +} + +impl GaugeOps for AtomicGauge { + fn increase(&self, value: u64) { + let _ = self + .atomic() + .fetch_update(Ordering::Relaxed, Ordering::Relaxed, |current| { + Some(current.saturating_add(value)) + }); + } + + fn decrease(&self, value: u64) { + let _ = self + .atomic() + .fetch_update(Ordering::Relaxed, Ordering::Relaxed, |current| { + Some(current.saturating_sub(value)) + }); + } + + fn absolute(&self, value: u64) { + self.atomic().store(value, Ordering::Relaxed); + } +} pub struct FoyerHybridCache { inner: foyer::HybridCache, + metrics: Option, } impl FoyerHybridCache { pub fn new_with_cache(cache: foyer::HybridCache) -> Self { - Self { inner: cache } + Self { + inner: cache, + metrics: None, + } + } + + /// Build a hybrid cache with retained block-engine accounting. + pub fn new_with_cache_and_metrics( + cache: foyer::HybridCache, + metrics: FoyerHybridCacheMetrics, + ) -> Self { + Self { + inner: cache, + metrics: Some(metrics), + } } } @@ -130,6 +345,21 @@ impl DbCache for FoyerHybridCache { 0 } + fn usage_snapshot(&self) -> DbCacheUsageSnapshot { + DbCacheUsageSnapshot { + memory: CacheUsageSnapshot::Ready { + used_bytes: self.inner.memory().usage() as u64, + capacity_bytes: Some(self.inner.memory().capacity() as u64), + }, + disk: self + .metrics + .as_ref() + .map_or(CacheUsageSnapshot::Unavailable, |metrics| { + metrics.disk_usage_snapshot() + }), + } + } + async fn close(&self) -> Result<(), crate::Error> { let memory_bytes = self.inner.memory().usage(); info!( @@ -203,8 +433,11 @@ impl FoyerHybridCache { #[cfg(test)] mod tests { - use crate::db_cache::foyer_hybrid::FoyerHybridCache; - use crate::db_cache::{CachedEntry, CachedKey, DbCache}; + use super::{ + FoyerCacheMetricsRegistry, FoyerHybridCache, FoyerHybridCacheMetrics, FOYER_BLOCK_GAUGE, + FOYER_BLOCK_SIZE_GAUGE, + }; + use crate::db_cache::{CacheUsageSnapshot, CachedEntry, CachedKey, DbCache}; use crate::db_state::SsTableId; use crate::format::sst::BlockBuilder; use foyer::{ @@ -217,6 +450,50 @@ mod tests { const SST_ID: SsTableId = SsTableId::Wal(123); + #[test] + fn tracks_pinned_foyer_block_metrics() { + use mixtrics::metrics::RegistryOps; + use std::borrow::Cow; + + let metrics = FoyerHybridCacheMetrics::new(); + let registry = FoyerCacheMetricsRegistry { + values: Arc::clone(&metrics.values), + }; + let block_states = registry.register_gauge_vec( + Cow::Borrowed(FOYER_BLOCK_GAUGE), + Cow::Borrowed(""), + &["name", "type"], + ); + block_states + .gauge(&[Cow::Borrowed("test"), Cow::Borrowed("clean")]) + .absolute(5); + block_states + .gauge(&[Cow::Borrowed("test"), Cow::Borrowed("writing")]) + .absolute(1); + block_states + .gauge(&[Cow::Borrowed("test"), Cow::Borrowed("evictable")]) + .absolute(2); + block_states + .gauge(&[Cow::Borrowed("test"), Cow::Borrowed("reclaiming")]) + .absolute(1); + registry + .register_gauge_vec( + Cow::Borrowed(FOYER_BLOCK_SIZE_GAUGE), + Cow::Borrowed(""), + &["name"], + ) + .gauge(&[Cow::Borrowed("test")]) + .absolute(4096); + + assert_eq!( + metrics.disk_usage_snapshot(), + CacheUsageSnapshot::Ready { + used_bytes: 4 * 4096, + capacity_bytes: Some(9 * 4096), + } + ); + } + #[tokio::test] async fn test_hybrid_cache() { let (cache, _dir) = setup().await; diff --git a/slatedb/src/db_cache/mod.rs b/slatedb/src/db_cache/mod.rs index f8dd48fa89..972e9757e3 100644 --- a/slatedb/src/db_cache/mod.rs +++ b/slatedb/src/db_cache/mod.rs @@ -45,6 +45,93 @@ pub const DEFAULT_MAX_CAPACITY: u64 = 64 * 1024 * 1024; pub const DEFAULT_BLOCK_CACHE_CAPACITY: u64 = 512 * 1024 * 1024; pub const DEFAULT_META_CACHE_CAPACITY: u64 = 128 * 1024 * 1024; +/// Point-in-time byte accounting for one cache tier. +/// +/// `Unavailable` is deliberately distinct from an empty ready cache. Callers +/// must not turn an unavailable measurement into zero. +#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)] +pub enum CacheUsageSnapshot { + /// The cache tier is not configured. + Disabled, + /// This cache implementation cannot currently provide byte accounting. + #[default] + Unavailable, + /// The cache is discovering its current contents. + Initializing { + /// Configured capacity, if the cache is bounded. + capacity_bytes: Option, + }, + /// The cache has a usable byte measurement. + Ready { + /// Bytes charged by the cache implementation. + used_bytes: u64, + /// Configured or usable capacity, absent for an unbounded cache. + capacity_bytes: Option, + }, +} + +impl CacheUsageSnapshot { + fn sum_enabled<'a>( + snapshots: impl Iterator, + ) -> CacheUsageSnapshot { + let mut found_enabled = false; + let mut initializing = false; + let mut used_bytes = 0_u64; + let mut capacity_bytes = Some(0_u64); + for snapshot in snapshots { + let (child_used, child_capacity) = match snapshot { + Self::Disabled => continue, + Self::Unavailable => return Self::Unavailable, + Self::Initializing { capacity_bytes } => { + found_enabled = true; + initializing = true; + (0, capacity_bytes) + } + Self::Ready { + used_bytes, + capacity_bytes, + } => { + found_enabled = true; + (*used_bytes, capacity_bytes) + } + }; + used_bytes = used_bytes.saturating_add(child_used); + capacity_bytes = match (capacity_bytes, child_capacity) { + (Some(total), Some(child)) => Some(total.saturating_add(*child)), + _ => None, + }; + } + if !found_enabled { + Self::Disabled + } else if initializing { + Self::Initializing { capacity_bytes } + } else { + Self::Ready { + used_bytes, + capacity_bytes, + } + } + } +} + +/// Point-in-time accounting for SlateDB's memory and optional disk cache tiers. +#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)] +pub struct DbCacheUsageSnapshot { + /// Cache-accounted resident memory. + pub memory: CacheUsageSnapshot, + /// Cache-accounted local disk. + pub disk: CacheUsageSnapshot, +} + +/// Point-in-time accounting for all cache tiers owned by one SlateDB handle. +#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)] +pub struct SlateDbCacheUsageSnapshot { + /// Block and metadata cache accounting. + pub db_cache: DbCacheUsageSnapshot, + /// Optional local object-store file cache accounting. + pub object_store: CacheUsageSnapshot, +} + /// Atomic counter to generate unique scope IDs for `DbCacheWrapper` instances. /// Scope `0` belongs exclusively to cache keys serialized before scoping was /// introduced, so live wrappers start at `1` and can never alias those entries. @@ -171,6 +258,11 @@ pub trait DbCache: Send + Sync { #[allow(dead_code)] fn entry_count(&self) -> u64; + /// Return current cache byte accounting without performing I/O. + fn usage_snapshot(&self) -> DbCacheUsageSnapshot { + DbCacheUsageSnapshot::default() + } + /// Gracefully close the cache, flushing any in-memory state to disk. /// /// Implementations backed by hybrid (memory + disk) caches should use @@ -577,6 +669,20 @@ impl DbCache for SplitCache { + self.meta_cache.as_ref().map_or(0, |c| c.entry_count()) } + fn usage_snapshot(&self) -> DbCacheUsageSnapshot { + let snapshots = [self.block_cache.as_ref(), self.meta_cache.as_ref()] + .into_iter() + .flatten() + .map(|cache| cache.usage_snapshot()) + .collect::>(); + DbCacheUsageSnapshot { + memory: CacheUsageSnapshot::sum_enabled( + snapshots.iter().map(|snapshot| &snapshot.memory), + ), + disk: CacheUsageSnapshot::sum_enabled(snapshots.iter().map(|snapshot| &snapshot.disk)), + } + } + async fn close(&self) -> Result<(), crate::Error> { if let Some(ref cache) = self.block_cache { cache.close().await?; @@ -830,6 +936,10 @@ impl DbCache for DbCacheWrapper { self.cache.entry_count() } + fn usage_snapshot(&self) -> DbCacheUsageSnapshot { + self.cache.usage_snapshot() + } + async fn close(&self) -> Result<(), crate::Error> { self.cache.close().await } @@ -933,6 +1043,10 @@ impl DbCache for UnownedDbCache { self.inner.entry_count() } + fn usage_snapshot(&self) -> DbCacheUsageSnapshot { + self.inner.usage_snapshot() + } + /// The point of this type: never propagate close to a cache we don't own. async fn close(&self) -> Result<(), crate::Error> { Ok(()) @@ -1190,7 +1304,9 @@ pub(crate) mod test_utils { #[cfg(test)] mod tests { - use crate::db_cache::{CachedEntry, CachedKey, DbCache, DbCacheWrapper, SplitCache}; + use crate::db_cache::{ + CacheUsageSnapshot, CachedEntry, CachedKey, DbCache, DbCacheWrapper, SplitCache, + }; use crate::db_state::SsTableId; use crate::filter_policy::{BloomFilterPolicy, FilterPolicy, NamedFilter}; use crate::format::sst::BlockBuilder; @@ -1199,6 +1315,58 @@ mod tests { use crate::flatbuffer_types::test_utils::assert_index_clamped; use crate::db_cache::test_utils::TestCache; + + #[test] + fn usage_snapshot_sum_preserves_typed_states() { + let snapshots = [ + CacheUsageSnapshot::Disabled, + CacheUsageSnapshot::Ready { + used_bytes: 7, + capacity_bytes: Some(10), + }, + CacheUsageSnapshot::Initializing { + capacity_bytes: Some(20), + }, + ]; + assert_eq!( + CacheUsageSnapshot::sum_enabled(snapshots.iter()), + CacheUsageSnapshot::Initializing { + capacity_bytes: Some(30), + } + ); + assert_eq!( + CacheUsageSnapshot::sum_enabled( + [ + CacheUsageSnapshot::Ready { + used_bytes: 12, + capacity_bytes: None, + }, + CacheUsageSnapshot::Ready { + used_bytes: 4, + capacity_bytes: Some(8), + }, + ] + .iter() + ), + CacheUsageSnapshot::Ready { + used_bytes: 16, + capacity_bytes: None, + } + ); + assert_eq!( + CacheUsageSnapshot::sum_enabled( + [ + CacheUsageSnapshot::Ready { + used_bytes: 1, + capacity_bytes: Some(1), + }, + CacheUsageSnapshot::Unavailable, + ] + .iter() + ), + CacheUsageSnapshot::Unavailable + ); + } use crate::format::sst::{EncodedSsTable, SsTableFormat}; use crate::test_utils::build_test_sst; use crate::types::{RowEntry, ValueDeletable}; diff --git a/slatedb/src/db_reader.rs b/slatedb/src/db_reader.rs index 2c0da334f0..5a32009994 100644 --- a/slatedb/src/db_reader.rs +++ b/slatedb/src/db_reader.rs @@ -121,6 +121,7 @@ impl WalReplayEnd { pub struct DbReader { inner: Arc, task_executor: MessageHandlerExecutor, + pub(crate) object_store_cache: Option>, } pub(crate) struct DbReaderInner { @@ -1540,9 +1541,34 @@ impl DbReader { Ok(Self { inner, task_executor, + object_store_cache: None, }) } + /// Returns a synchronous point-in-time cache accounting snapshot. + /// + /// This method reads only in-memory counters and never performs object-store + /// or filesystem I/O. + pub fn cache_usage_snapshot(&self) -> crate::db_cache::SlateDbCacheUsageSnapshot { + let db_cache = self.inner.table_store.cache().map_or_else( + || crate::db_cache::DbCacheUsageSnapshot { + memory: crate::db_cache::CacheUsageSnapshot::Disabled, + disk: crate::db_cache::CacheUsageSnapshot::Disabled, + }, + |cache| cache.usage_snapshot(), + ); + let object_store = self + .object_store_cache + .as_ref() + .map_or(crate::db_cache::CacheUsageSnapshot::Disabled, |cache| { + cache.usage_snapshot() + }); + crate::db_cache::SlateDbCacheUsageSnapshot { + db_cache, + object_store, + } + } + /// Get a value from the database with default read options. /// /// The `Bytes` object returned contains a slice of an entire From 1ec9fc28a4f1c0514c53a46771cc767fd2caf7b7 Mon Sep 17 00:00:00 2001 From: xav-db Date: Wed, 29 Jul 2026 13:25:22 +0100 Subject: [PATCH 56/65] Update SlateDB dependency lockfile --- Cargo.lock | 1 + 1 file changed, 1 insertion(+) diff --git a/Cargo.lock b/Cargo.lock index c027072970..18e335f2e3 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -3239,6 +3239,7 @@ dependencies = [ "log", "lru", "lz4_flex", + "mixtrics", "moka", "object_store", "ouroboros", From fd0a6cbb6a9c6c0a96d085857d42f755d6ac7283 Mon Sep 17 00:00:00 2001 From: xav-db Date: Wed, 29 Jul 2026 14:01:16 +0100 Subject: [PATCH 57/65] Keep unbounded cache accounting synchronous --- slatedb/src/cached_object_store/storage_fs.rs | 211 +++++++++++++----- 1 file changed, 159 insertions(+), 52 deletions(-) diff --git a/slatedb/src/cached_object_store/storage_fs.rs b/slatedb/src/cached_object_store/storage_fs.rs index ea1eb383b1..87de02df96 100644 --- a/slatedb/src/cached_object_store/storage_fs.rs +++ b/slatedb/src/cached_object_store/storage_fs.rs @@ -232,18 +232,17 @@ pub(crate) struct FsCacheEntry { impl FsCacheEntry { async fn atomic_write(&self, path: std::path::PathBuf, buf: Bytes) -> object_store::Result<()> { let tmp_path = path.with_extension(format!("_tmp{}", self.make_rand_suffix())); + let bytes = buf.len(); - // try triggering evict before writing - if let Some(evictor) = &self.evictor { + let Some(write_reservation) = (if let Some(evictor) = &self.evictor { // If the evictor is backpressured, skip this cache write to avoid // stalling foreground PUTs. Cache writes are best-effort. - if !evictor - .track_entry_accessed(path.clone(), EntryAccess::Write(buf.len())) - .await - { - return Ok(()); - } - } + evictor.reserve_entry_access() + } else { + Some(WriteReservation::Untracked) + }) else { + return Ok(()); + }; // Spawn a blocking task and do synchronous I/O rather than use the tokio async apis. // Under the hood, on linux systems , tokio itself spawns a blocking task for each call to @@ -279,6 +278,10 @@ impl FsCacheEntry { .await? .map_err(wrap_io_err)?; + if let WriteReservation::Reserved(permit) = write_reservation { + permit.send((invalidate_path.clone(), EntryAccess::Write(bytes))); + } + // The rename replaced the file at `path`, so any previously cached // handle now points to the old (unlinked) inode. Invalidate it so // the next read opens the new file. @@ -525,9 +528,13 @@ impl LocalCacheEntry for FsCacheEntry { }; if let Some(evictor) = &self.evictor { - evictor - .track_entry_accessed(path, EntryAccess::Delete) - .await; + if evictor.max_cache_size_bytes.is_some() { + evictor + .track_entry_accessed(path, EntryAccess::Delete) + .await; + } else { + evictor.inner.delete_entry(path).await; + } } else { delete_cache_entry(path, self.file_handle_cache.clone()).await; } @@ -541,6 +548,11 @@ enum EntryAccess { } type FsCacheEvictorWork = (std::path::PathBuf, EntryAccess); + +enum WriteReservation<'a> { + Untracked, + Reserved(tokio::sync::mpsc::Permit<'a, FsCacheEvictorWork>), +} // Minimum time between aggregated "evictor queue is full" warnings. const QUEUE_FULL_LOG_INTERVAL_MS: i64 = 30_000; @@ -549,7 +561,6 @@ const QUEUE_FULL_LOG_INTERVAL_MS: i64 = 30_000; /// is added. #[derive(Debug)] struct FsCacheEvictor { - root_folder: std::path::PathBuf, max_cache_size_bytes: Option, scan_interval: Option, tx: tokio::sync::mpsc::Sender, @@ -559,12 +570,10 @@ struct FsCacheEvictor { last_queue_full_log_ms: AtomicI64, background_evict_handle: OnceCell>, background_scan_handle: OnceCell>, - stats: Arc, system_clock: Arc, - rand: Arc, - file_handle_cache: FileHandleCache, usage: Arc, reconcile_notify: Arc, + inner: Arc, } #[derive(Debug)] @@ -599,8 +608,20 @@ impl FsCacheEvictor { file_handle_cache: FileHandleCache, ) -> Self { let (tx, rx) = tokio::sync::mpsc::channel(100); + let usage = Arc::new(FsCacheUsage { + initialized: AtomicBool::new(false), + used_bytes: AtomicU64::new(0), + capacity_bytes: max_cache_size_bytes.map(|bytes| bytes as u64), + }); + let inner = Arc::new(FsCacheEvictorInner::new_with_usage( + root_folder.clone(), + max_cache_size_bytes, + stats.clone(), + rand.clone(), + file_handle_cache.clone(), + usage.clone(), + )); Self { - root_folder, scan_interval, max_cache_size_bytes, tx, @@ -610,28 +631,15 @@ impl FsCacheEvictor { last_queue_full_log_ms: AtomicI64::new(i64::MIN), background_evict_handle: OnceCell::new(), background_scan_handle: OnceCell::new(), - stats, system_clock, - rand, - file_handle_cache, - usage: Arc::new(FsCacheUsage { - initialized: AtomicBool::new(false), - used_bytes: AtomicU64::new(0), - capacity_bytes: max_cache_size_bytes.map(|bytes| bytes as u64), - }), + usage, reconcile_notify: Arc::new(Notify::new()), + inner, } } async fn start(&self) { - let inner = Arc::new(FsCacheEvictorInner::new_with_usage( - self.root_folder.clone(), - self.max_cache_size_bytes, - self.stats.clone(), - self.rand.clone(), - self.file_handle_cache.clone(), - self.usage.clone(), - )); + let inner = self.inner.clone(); let guard = self.rx.lock(); let rx = guard.await.take().expect("evictor already started"); @@ -667,6 +675,49 @@ impl FsCacheEvictor { self.usage.snapshot() } + fn reserve_entry_access(&self) -> Option> { + if !self.started() { + return Some(WriteReservation::Untracked); + } + match self.tx.try_reserve() { + Ok(permit) => Some(WriteReservation::Reserved(permit)), + Err(tokio::sync::mpsc::error::TrySendError::Full(_)) => { + self.mark_accounting_dirty(); + self.record_queue_full(); + None + } + Err(tokio::sync::mpsc::error::TrySendError::Closed(_)) => { + self.mark_accounting_dirty(); + None + } + } + } + + fn mark_accounting_dirty(&self) { + self.usage.initialized.store(false, Ordering::Release); + self.reconcile_notify.notify_one(); + } + + fn record_queue_full(&self) { + self.queue_full_count.fetch_add(1, Ordering::AcqRel); + let now_ms = self.system_clock.now().timestamp_millis(); + let last_log_ms = self.last_queue_full_log_ms.load(Ordering::Acquire); + if now_ms.saturating_sub(last_log_ms) < QUEUE_FULL_LOG_INTERVAL_MS { + return; + } + if self + .last_queue_full_log_ms + .compare_exchange(last_log_ms, now_ms, Ordering::AcqRel, Ordering::Acquire) + .is_ok() + { + let queue_full_count = self.queue_full_count.swap(0, Ordering::AcqRel); + warn!( + "evictor queue skipped cache write/access event because it was full {} times in the last 30s", + queue_full_count, + ); + } + } + async fn background_evict( inner: Arc, mut rx: tokio::sync::mpsc::Receiver, @@ -728,28 +779,12 @@ impl FsCacheEvictor { match self.tx.try_send((path, access)) { Ok(()) => true, Err(tokio::sync::mpsc::error::TrySendError::Full(_)) => { - self.usage.initialized.store(false, Ordering::Release); - self.reconcile_notify.notify_one(); - self.queue_full_count.fetch_add(1, Ordering::AcqRel); - let now_ms = self.system_clock.now().timestamp_millis(); - let last_log_ms = self.last_queue_full_log_ms.load(Ordering::Acquire); - if now_ms.saturating_sub(last_log_ms) >= QUEUE_FULL_LOG_INTERVAL_MS - && self - .last_queue_full_log_ms - .compare_exchange(last_log_ms, now_ms, Ordering::AcqRel, Ordering::Acquire) - .is_ok() - { - let queue_full_count = self.queue_full_count.swap(0, Ordering::AcqRel); - warn!( - "evictor queue skipped cache write/access event because it was full {} times in the last 30s", - queue_full_count, - ); - } + self.mark_accounting_dirty(); + self.record_queue_full(); false } Err(tokio::sync::mpsc::error::TrySendError::Closed(_)) => { - self.usage.initialized.store(false, Ordering::Release); - self.reconcile_notify.notify_one(); + self.mark_accounting_dirty(); false } } @@ -1378,6 +1413,15 @@ mod tests { assert!(accepted); } + evictor.usage.initialized.store(true, Ordering::Release); + assert!(evictor.reserve_entry_access().is_none()); + assert_eq!( + evictor.usage_snapshot(), + CacheUsageSnapshot::Initializing { + capacity_bytes: Some(1024), + } + ); + let accepted = evictor .track_entry_accessed(std::path::PathBuf::from("overflow"), EntryAccess::Write(1)) .await; @@ -1517,6 +1561,69 @@ mod tests { ); } + #[tokio::test] + async fn unbounded_usage_tracks_successful_write_overwrite_and_delete() { + let temp_dir = tempfile::Builder::new() + .prefix("objstore_cache_usage_updates_") + .tempdir() + .unwrap(); + let recorder = slatedb_common::metrics::MetricsRecorderHelper::noop(); + let storage = FsCacheStorage::new( + temp_dir.path().to_path_buf(), + None, + None, + Arc::new(CachedObjectStoreStats::new(&recorder)), + Arc::new(DefaultSystemClock::new()), + Arc::new(DbRand::default()), + 10, + ); + storage.start_evictor().await; + tokio::time::timeout(Duration::from_secs(1), async { + while !matches!( + storage.usage_snapshot(), + CacheUsageSnapshot::Ready { used_bytes: 0, .. } + ) { + tokio::task::yield_now().await; + } + }) + .await + .unwrap(); + + let entry = storage.entry(&Path::from("cached/object"), 1024); + entry.save_part(0, Bytes::from(vec![1; 17])).await.unwrap(); + tokio::time::timeout(Duration::from_secs(1), async { + while !matches!( + storage.usage_snapshot(), + CacheUsageSnapshot::Ready { used_bytes: 17, .. } + ) { + tokio::task::yield_now().await; + } + }) + .await + .unwrap(); + + entry.save_part(0, Bytes::from(vec![2; 29])).await.unwrap(); + tokio::time::timeout(Duration::from_secs(1), async { + while !matches!( + storage.usage_snapshot(), + CacheUsageSnapshot::Ready { used_bytes: 29, .. } + ) { + tokio::task::yield_now().await; + } + }) + .await + .unwrap(); + + entry.delete().await; + assert_eq!( + storage.usage_snapshot(), + CacheUsageSnapshot::Ready { + used_bytes: 0, + capacity_bytes: None, + } + ); + } + #[rstest::rstest] // Basic case: 2 keys, nothing picked, no exclusion #[case(&[0, 1], &[], None, &[0, 1])] From 3d2e4af7b6f723b7b26b126ca0da5cdb4f4de6bc Mon Sep 17 00:00:00 2001 From: xav-db Date: Wed, 29 Jul 2026 13:40:16 +0100 Subject: [PATCH 58/65] Add typed database-missing reader error --- slatedb/src/db_reader.rs | 27 +++++++++++++++++++- slatedb/src/error.rs | 53 +++++++++++++++++++++++++++++++++++++++- slatedb/src/lib.rs | 2 +- 3 files changed, 79 insertions(+), 3 deletions(-) diff --git a/slatedb/src/db_reader.rs b/slatedb/src/db_reader.rs index 5a32009994..89c4df7f12 100644 --- a/slatedb/src/db_reader.rs +++ b/slatedb/src/db_reader.rs @@ -1501,7 +1501,13 @@ impl DbReader { Self::validate_options(mode, &options)?; let manifest = - StoredManifest::load(Arc::clone(&manifest_store), system_clock.clone()).await?; + match StoredManifest::load(Arc::clone(&manifest_store), system_clock.clone()).await { + Ok(manifest) => manifest, + Err(SlateDBError::LatestTransactionalObjectVersionMissing) => { + return Err(SlateDBError::DatabaseMissing); + } + Err(error) => return Err(error), + }; if !manifest.db_state().initialized { return Err(SlateDBError::InvalidDBState); } @@ -2239,6 +2245,25 @@ mod tests { reader.close().await.unwrap(); } + #[tokio::test] + async fn empty_database_reader_returns_typed_database_missing() { + let object_store: Arc = Arc::new(InMemory::new()); + let error = match DbReader::open( + "/tmp/test_reader_database_missing", + object_store, + None, + DbReaderOptions::default(), + ) + .await + { + Ok(_) => panic!("empty object store must not open a reader"), + Err(error) => error, + }; + + assert_eq!(error.kind(), crate::ErrorKind::Data); + assert_eq!(error.code(), Some(crate::ErrorCode::DatabaseMissing)); + } + #[tokio::test] async fn should_return_current_versioned_manifest() { let object_store: Arc = Arc::new(InMemory::new()); diff --git a/slatedb/src/error.rs b/slatedb/src/error.rs index fc88d3ff13..d7fd2f259f 100644 --- a/slatedb/src/error.rs +++ b/slatedb/src/error.rs @@ -8,7 +8,7 @@ use uuid::Uuid; use crate::bytes_range::BytesRange; use crate::error::SlateDBError::{ - LatestTransactionalObjectVersionMissing, TransactionalObjectVersionExists, + DatabaseMissing, LatestTransactionalObjectVersionMissing, TransactionalObjectVersionExists, }; use crate::merge_operator::MergeOperatorError; use slatedb_txn_obj::TransactionalObjectError; @@ -55,6 +55,9 @@ pub(crate) enum SlateDBError { #[error("failed to find latest transactional object (e.g. manifest) version")] LatestTransactionalObjectVersionMissing, + #[error("database does not exist")] + DatabaseMissing, + #[error("generic transactional object (e.g. manifest) error {0:?}")] TransactionalObjectError(#[from] Arc), @@ -496,6 +499,15 @@ pub enum ErrorKind { Internal, } +/// Stable machine-readable detail for public errors that require more precise +/// handling than [`ErrorKind`]. +#[non_exhaustive] +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum ErrorCode { + /// A reader cannot open because no database manifest exists. + DatabaseMissing, +} + impl From for CloseReason { fn from(kind: ErrorKind) -> Self { match kind { @@ -548,6 +560,7 @@ pub enum RetryReason { pub struct Error { msg: String, kind: ErrorKind, + code: Option, source: Option, } @@ -575,6 +588,7 @@ impl Error { Self { msg, kind: ErrorKind::Transaction, + code: None, source: None, } } @@ -584,6 +598,7 @@ impl Error { Self { msg, kind: ErrorKind::Closed(reason), + code: None, source: None, } } @@ -593,6 +608,7 @@ impl Error { Self { msg, kind: ErrorKind::Unavailable, + code: None, source: None, } } @@ -602,6 +618,7 @@ impl Error { Self { msg, kind: ErrorKind::Invalid, + code: None, source: None, } } @@ -611,6 +628,7 @@ impl Error { Self { msg, kind: ErrorKind::Data, + code: None, source: None, } } @@ -620,6 +638,7 @@ impl Error { Self { msg, kind: ErrorKind::Internal, + code: None, source: None, } } @@ -630,10 +649,20 @@ impl Error { self } + fn with_code(mut self, code: ErrorCode) -> Self { + self.code = Some(code); + self + } + /// Returns the error kind. pub fn kind(&self) -> ErrorKind { self.kind } + + /// Returns stable machine-readable detail when one is available. + pub fn code(&self) -> Option { + self.code + } } impl From for Error { @@ -727,6 +756,7 @@ impl From for Error { SlateDBError::InvalidVersion { .. } => Error::data(msg), SlateDBError::ManifestMissing(_) => Error::data(msg), LatestTransactionalObjectVersionMissing => Error::data(msg), + DatabaseMissing => Error::data(msg).with_code(ErrorCode::DatabaseMissing), TransactionalObjectVersionExists => Error::data(msg), SlateDBError::InvalidTransactionalObjectState => Error::data(msg), SlateDBError::EmptyManifest => Error::data(msg), @@ -795,4 +825,25 @@ mod tests { assert_eq!(public_err.kind(), ErrorKind::Unavailable); } + + #[test] + fn database_missing_has_stable_code_without_changing_broad_kind() { + let public_err = Error::from(SlateDBError::DatabaseMissing); + + assert_eq!(public_err.kind(), ErrorKind::Data); + assert_eq!(public_err.code(), Some(ErrorCode::DatabaseMissing)); + } + + #[test] + fn other_data_errors_do_not_claim_database_is_missing() { + for err in [ + SlateDBError::LatestTransactionalObjectVersionMissing, + SlateDBError::ManifestMissing(7), + SlateDBError::InvalidDBState, + ] { + let public_err = Error::from(err); + assert_eq!(public_err.kind(), ErrorKind::Data); + assert_eq!(public_err.code(), None); + } + } } diff --git a/slatedb/src/lib.rs b/slatedb/src/lib.rs index 875ea32240..d86df700a3 100644 --- a/slatedb/src/lib.rs +++ b/slatedb/src/lib.rs @@ -54,7 +54,7 @@ pub use db_iter::{DbIterator, DbRecencyIterator}; pub use db_reader::{DbReader, DbReaderMode}; pub use db_snapshot::DbSnapshot; pub use db_transaction::DbTransaction; -pub use error::{CloseReason, Error, ErrorKind}; +pub use error::{CloseReason, Error, ErrorCode, ErrorKind}; pub use filter::BloomFilter; pub use filter_policy::{ BloomFilterPolicy, Filter, FilterBuilder, FilterContext, FilterPolicy, FilterQuery, From 36b34656e4dde33f5a078fc8bcc821994c8a8378 Mon Sep 17 00:00:00 2001 From: xav-db Date: Wed, 29 Jul 2026 14:47:51 +0100 Subject: [PATCH 59/65] Update missing database binding tests --- bindings/node/tests/reader.test.mjs | 2 +- bindings/python/tests/test_reader.py | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/bindings/node/tests/reader.test.mjs b/bindings/node/tests/reader.test.mjs index ca01928414..180b85f5e0 100644 --- a/bindings/node/tests/reader.test.mjs +++ b/bindings/node/tests/reader.test.mjs @@ -67,7 +67,7 @@ test("reader build fails when database is missing", async (t) => { const builder = cleanup.track(new DbReaderBuilder(TEST_DB_PATH, store), { shutdown: false }); const error = await expectError(() => builder.build(), ErrorData); - assert.match(error.message, /failed to find latest transactional object/); + assert.match(error.message, /database does not exist/); }); test("reader point reads", async (t) => { diff --git a/bindings/python/tests/test_reader.py b/bindings/python/tests/test_reader.py index 5567c1b95b..f3fde22047 100644 --- a/bindings/python/tests/test_reader.py +++ b/bindings/python/tests/test_reader.py @@ -52,7 +52,7 @@ async def test_reader_build_fails_when_database_is_missing() -> None: with pytest.raises(Error.Data) as exc: await builder.build() - assert "failed to find latest transactional object" in exc.value.message + assert "database does not exist" in exc.value.message @pytest.mark.asyncio From 0fbf6c05e4d0951a36b547f926679febf90c9a9c Mon Sep 17 00:00:00 2001 From: Matthew Sanetra <41018997+matthewsanetra@users.noreply.github.com> Date: Fri, 7 Aug 2026 16:23:00 +0100 Subject: [PATCH 60/65] Finish v0.16 reader API adaptation --- slatedb/benches/db_reader_scaling.rs | 4 +-- slatedb/src/db_reader.rs | 37 +++++++--------------------- slatedb/src/db_transaction.rs | 5 +--- slatedb/src/reader.rs | 4 +-- 4 files changed, 13 insertions(+), 37 deletions(-) diff --git a/slatedb/benches/db_reader_scaling.rs b/slatedb/benches/db_reader_scaling.rs index 278cb4ea06..c55c23c4e8 100644 --- a/slatedb/benches/db_reader_scaling.rs +++ b/slatedb/benches/db_reader_scaling.rs @@ -26,7 +26,7 @@ use slatedb::config::{ Settings, WriteOptions, }; use slatedb::instrumented_object_store_stats; -use slatedb::{Db, DbReader, DbSnapshot, PrefixExtractor, PrefixTarget}; +use slatedb::{Db, DbReader, DbReaderMode, DbSnapshot, PrefixExtractor, PrefixTarget}; use slatedb_common::metrics::{lookup_metric, lookup_metric_with_labels, DefaultMetricsRecorder}; use slatedb_common::{MockSystemClock, SystemClock}; use tokio::sync::Barrier; @@ -1151,7 +1151,7 @@ async fn benchmark_fixed_reader_open(scales: &[usize], full: bool) { let object_store: Arc = store.clone(); let start = Instant::now(); let reader = DbReader::builder(path.as_str(), object_store) - .with_checkpoint_id(checkpoint.id) + .with_reader_mode(DbReaderMode::Checkpoint(checkpoint.id)) .with_options(quiet_reader_options(1)) .with_metrics_recorder(recorder.clone()) .with_db_cache_disabled() diff --git a/slatedb/src/db_reader.rs b/slatedb/src/db_reader.rs index 89c4df7f12..6ab8e7e25e 100644 --- a/slatedb/src/db_reader.rs +++ b/slatedb/src/db_reader.rs @@ -180,11 +180,7 @@ impl Drop for ReaderGenerationPermit { } impl ReaderGeneration { - fn new( - manifest_id: u64, - checkpoint: Option, - manifest: Manifest, - ) -> Arc { + fn new(manifest_id: u64, checkpoint: Option, manifest: Manifest) -> Arc { Arc::new(Self { manifest_id, checkpoint: checkpoint.map(RwLock::new), @@ -195,7 +191,9 @@ impl ReaderGeneration { } fn checkpoint(&self) -> Option { - self.checkpoint.as_ref().map(|checkpoint| checkpoint.read().clone()) + self.checkpoint + .as_ref() + .map(|checkpoint| checkpoint.read().clone()) } fn managed_checkpoint(&self) -> &RwLock { @@ -585,9 +583,7 @@ impl DbReaderInner { .await } - pub(crate) async fn snapshot_multi_get_key_value_with_options< - K: AsRef<[u8]> + Send + Sync, - >( + pub(crate) async fn snapshot_multi_get_key_value_with_options + Send + Sync>( &self, state: Arc, max_seq: u64, @@ -1124,10 +1120,7 @@ impl ManifestPoller { .id; generations.insert(checkpoint_id, Arc::downgrade(&generation)); } - let poller = Self { - inner, - generations, - }; + let poller = Self { inner, generations }; poller.report_active_checkpoints(); poller } @@ -2251,7 +2244,7 @@ mod tests { let error = match DbReader::open( "/tmp/test_reader_database_missing", object_store, - None, + DbReaderMode::ManagedCheckpoint, DbReaderOptions::default(), ) .await @@ -2990,13 +2983,7 @@ mod tests { ) .await .unwrap(); - let reader_checkpoint_id = inner - .state - .read() - .generation - .checkpoint() - .unwrap() - .id; + let reader_checkpoint_id = inner.state.read().generation.checkpoint().unwrap().id; // Simulate the writer's GC reaping the expired checkpoint. let mut stored_manifest = StoredManifest::load(Arc::clone(&manifest_store), clock.clone()) @@ -3018,13 +3005,7 @@ mod tests { .unwrap(); // The reader should have replaced the reaped checkpoint with a new one. - let new_checkpoint_id = inner - .state - .read() - .generation - .checkpoint() - .unwrap() - .id; + let new_checkpoint_id = inner.state.read().generation.checkpoint().unwrap().id; assert_ne!(reader_checkpoint_id, new_checkpoint_id); let latest_manifest = manifest_store.read_latest_manifest().await.unwrap(); let checkpoints = &latest_manifest.manifest.core.checkpoints; diff --git a/slatedb/src/db_transaction.rs b/slatedb/src/db_transaction.rs index 34855d333b..79898e7f50 100644 --- a/slatedb/src/db_transaction.rs +++ b/slatedb/src/db_transaction.rs @@ -295,10 +295,7 @@ impl DbTransaction { ) .await .map_err(crate::Error::from)?; - for (key_idx, value) in reader_key_indices - .into_iter() - .zip(reader_values.into_iter()) - { + for (key_idx, value) in reader_key_indices.into_iter().zip(reader_values) { resolved[key_idx] = true; values[key_idx] = value; } diff --git a/slatedb/src/reader.rs b/slatedb/src/reader.rs index 777a68d00d..0926753c6f 100644 --- a/slatedb/src/reader.rs +++ b/slatedb/src/reader.rs @@ -1154,9 +1154,7 @@ mod tests { self.memtable.clone() } - fn imm_memtables( - &self, - ) -> Box> + Send + '_> { + fn imm_memtables(&self) -> Box> + Send + '_> { Box::new(self.imm_memtable.iter().cloned()) } From c37a70eb7800121f5c7f55ba20ebca04176eda51 Mon Sep 17 00:00:00 2001 From: Matthew Sanetra <41018997+matthewsanetra@users.noreply.github.com> Date: Fri, 14 Aug 2026 15:21:06 +0100 Subject: [PATCH 61/65] Add conflict-compatible transaction merges --- slatedb/src/batch.rs | 135 ++++++++++- slatedb/src/db_transaction.rs | 357 ++++++++++++++++++++++++++++- slatedb/src/ops.rs | 19 ++ slatedb/src/transaction_manager.rs | 303 +++++++++++++++++++----- 4 files changed, 754 insertions(+), 60 deletions(-) diff --git a/slatedb/src/batch.rs b/slatedb/src/batch.rs index 2be7bcdebd..2cb6b2f758 100644 --- a/slatedb/src/batch.rs +++ b/slatedb/src/batch.rs @@ -53,6 +53,18 @@ pub struct WriteBatch { pub(crate) ops: BTreeMap>, pub(crate) op_count: usize, pub(crate) has_merge_ops: bool, + /// Keys whose surviving operations are exclusively commutative merges. + /// + /// This is transaction conflict metadata only. It is never encoded into a + /// [`RowEntry`] or persisted by the WAL, memtable, or compaction paths. The + /// set is boxed and allocated lazily so ordinary write batches retain their + /// existing inline size and pay no allocation for this opt-in metadata. + commutative_merge_keys: Option>, +} + +#[derive(Clone, Debug, Default)] +struct CommutativeMergeKeys { + keys: HashSet, } impl Default for WriteBatch { @@ -146,6 +158,7 @@ impl WriteBatch { ops: BTreeMap::new(), op_count: 0, has_merge_ops: false, + commutative_merge_keys: None, } } @@ -224,6 +237,7 @@ impl WriteBatch { /// - if the value size is larger than u32::MAX pub fn put_bytes_with_options(&mut self, key: Bytes, value: Bytes, options: &PutOptions) { self.assert_kv(&key, &value); + self.remove_commutative_merge_key(&key); // put will overwrite the existing key so we can safely // remove all previous entries. @@ -250,16 +264,57 @@ impl WriteBatch { where K: AsRef<[u8]>, V: AsRef<[u8]>, + { + self.merge_with_conflict_kind(key, value, options, false); + } + + /// Merge a key-value pair whose operands are commutative. + /// + /// This remains crate-private because compatibility affects transaction + /// conflict detection, while public non-transactional batches do not run + /// write/write conflict checks. + pub(crate) fn merge_commutative(&mut self, key: K, value: V) + where + K: AsRef<[u8]>, + V: AsRef<[u8]>, + { + self.merge_with_conflict_kind(key, value, &MergeOptions::default(), true); + } + + fn merge_with_conflict_kind( + &mut self, + key: K, + value: V, + options: &MergeOptions, + commutative: bool, + ) where + K: AsRef<[u8]>, + V: AsRef<[u8]>, { self.assert_kv(&key, &value); - let key = key.as_ref(); + let key = Bytes::copy_from_slice(key.as_ref()); let value = value.as_ref(); let op = WriteOp::Merge(Bytes::copy_from_slice(value), options.clone()); - if let Some(ops) = self.ops.get_mut(key) { + let existing_operations_are_commutative = self.ops.contains_key(&key) + && self + .commutative_merge_keys + .as_ref() + .is_some_and(|keys| keys.keys.contains(&key)); + let is_first_operation = !self.ops.contains_key(&key); + if commutative && (is_first_operation || existing_operations_are_commutative) { + self.commutative_merge_keys + .get_or_insert_default() + .keys + .insert(key.clone()); + } else { + self.remove_commutative_merge_key(&key); + } + + if let Some(ops) = self.ops.get_mut(&key) { ops.push(op); } else { - self.ops.insert(Bytes::copy_from_slice(key), smallvec![op]); + self.ops.insert(key, smallvec![op]); } self.has_merge_ops = true; @@ -271,6 +326,7 @@ impl WriteBatch { self.assert_kv(&key, &[]); let key = Bytes::copy_from_slice(key.as_ref()); + self.remove_commutative_merge_key(&key); // delete will overwrite the existing key so we can safely // remove all previous entries. @@ -306,6 +362,23 @@ impl WriteBatch { self.ops.keys().cloned().collect() } + /// Returns whether every surviving operation for `key` is a commutative merge. + pub(crate) fn is_commutative_merge_key(&self, key: &[u8]) -> bool { + self.commutative_merge_keys + .as_ref() + .is_some_and(|keys| keys.keys.contains(key)) + } + + fn remove_commutative_merge_key(&mut self, key: &[u8]) { + let Some(keys) = self.commutative_merge_keys.as_mut() else { + return; + }; + keys.keys.remove(key); + if keys.keys.is_empty() { + self.commutative_merge_keys = None; + } + } + /// Converts a WriteBatch into a vector of RowEntry objects with /// seq and timestamp set, applying the merge operator to any /// mergeable entries. @@ -846,6 +919,62 @@ mod tests { assert_iterator(&mut iter, expected).await; } + #[test] + fn commutative_merge_classification_requires_only_commutative_operations() { + let mut only_commutative = WriteBatch::new(); + only_commutative.merge_commutative(b"key", b"first"); + only_commutative.merge_commutative(b"key", b"second"); + assert!(only_commutative.is_commutative_merge_key(b"key")); + assert!(only_commutative.commutative_merge_keys.is_some()); + + let mut commutative_then_ordinary = WriteBatch::new(); + commutative_then_ordinary.merge_commutative(b"key", b"first"); + commutative_then_ordinary.merge(b"key", b"second"); + assert!(!commutative_then_ordinary.is_commutative_merge_key(b"key")); + assert!(commutative_then_ordinary.commutative_merge_keys.is_none()); + + let mut ordinary_then_commutative = WriteBatch::new(); + ordinary_then_commutative.merge(b"key", b"first"); + ordinary_then_commutative.merge_commutative(b"key", b"second"); + assert!(!ordinary_then_commutative.is_commutative_merge_key(b"key")); + } + + #[test] + fn put_and_delete_make_commutative_merge_keys_exclusive() { + let mut put_after_merge = WriteBatch::new(); + put_after_merge.merge_commutative(b"key", b"merge"); + put_after_merge.put(b"key", b"put"); + assert!(!put_after_merge.is_commutative_merge_key(b"key")); + + let mut merge_after_put = WriteBatch::new(); + merge_after_put.put(b"key", b"put"); + merge_after_put.merge_commutative(b"key", b"merge"); + assert!(!merge_after_put.is_commutative_merge_key(b"key")); + + let mut delete_after_merge = WriteBatch::new(); + delete_after_merge.merge_commutative(b"key", b"merge"); + delete_after_merge.delete(b"key"); + assert!(!delete_after_merge.is_commutative_merge_key(b"key")); + + let mut merge_after_delete = WriteBatch::new(); + merge_after_delete.delete(b"key"); + merge_after_delete.merge_commutative(b"key", b"merge"); + assert!(!merge_after_delete.is_commutative_merge_key(b"key")); + } + + #[test] + fn commutative_merge_metadata_does_not_change_write_operations() { + let mut ordinary = WriteBatch::new(); + ordinary.merge(b"key", b"operand"); + + let mut commutative = WriteBatch::new(); + commutative.merge_commutative(b"key", b"operand"); + + assert_eq!(ordinary.ops, commutative.ops); + assert!(!ordinary.is_commutative_merge_key(b"key")); + assert!(commutative.is_commutative_merge_key(b"key")); + } + #[test] fn should_create_merge_operation_with_default_options() { // Given: an empty WriteBatch diff --git a/slatedb/src/db_transaction.rs b/slatedb/src/db_transaction.rs index 79898e7f50..ced0f32bf5 100644 --- a/slatedb/src/db_transaction.rs +++ b/slatedb/src/db_transaction.rs @@ -1,6 +1,6 @@ use bytes::Bytes; use parking_lot::{Mutex, RwLock}; -use std::collections::HashSet; +use std::collections::{HashMap, HashSet}; use std::sync::Arc; use uuid::Uuid; @@ -13,7 +13,7 @@ use crate::db_iter::{DbIterator, DbIteratorRangeTracker}; use crate::error::SlateDBError; use crate::iter::IterationOrder; use crate::reader::ScanContext; -use crate::transaction_manager::{IsolationLevel, TransactionManager}; +use crate::transaction_manager::{IsolationLevel, TransactionManager, TransactionWriteKind}; use crate::types::KeyValue; use crate::{DbReadOps, DbTransactionOps}; @@ -639,6 +639,123 @@ impl DbTransaction { self.merge_with_options(key, value, &MergeOptions::default()) } + /// Buffers a merge operand whose effect is independent of its ordering + /// relative to other commutative merge operands for the same key. + /// + /// This is an opt-in write/write conflict optimization for transactions + /// performing blind, merge-only updates. Unlike [`Self::merge`], concurrent + /// transactions using this method for the same key may both commit. Both + /// operands are retained and applied by the configured [`crate::MergeOperator`]. + /// + /// # Required algebraic contract + /// + /// Let `apply(base, operand)` represent the configured merge operator. For + /// every base value and every pair of operands that may concurrently target + /// the key, the caller must ensure: + /// + /// ```text + /// apply(apply(base, a), b) == apply(apply(base, b), a) + /// ``` + /// + /// The existing associativity requirements of [`crate::MergeOperator`] and + /// [`crate::MergeOperator::merge_batch`] also continue to apply. SlateDB + /// cannot inspect an application-defined merge operator or prove this + /// property. Idempotence is not required, so additive counter deltas are + /// valid. Do not use this method for order-sensitive operations. + /// + /// # Conflict and isolation guarantees + /// + /// This remains a tracked write and does not have the broad semantics of + /// [`Self::unmark_write`]. + /// + /// - A write/write conflict is omitted only when both transactions classify + /// their final operations for the key exclusively as commutative merges. + /// - A put, delete, ordinary merge, or mixture of operation kinds makes the + /// key exclusive and restores normal write/write conflict detection. + /// - Serializable point-read and range-read dependencies continue to see + /// this key as written and can still cause the transaction to abort. + /// - Reading the key before calling this method does not suppress that read + /// dependency, and conflicts involving other keys are unaffected. + /// + /// The operand remains part of the transaction's ordinary atomic write + /// batch. Commit ordering, durability, snapshot visibility, merge-operator + /// execution, and error propagation are unchanged. The commutative marker + /// is transaction conflict metadata only and is not stored on disk. + /// + /// Custom [`MergeOptions`] are intentionally unsupported because TTL or + /// other order-sensitive options would require a separate compatibility + /// contract. + /// + /// # Errors + /// + /// Returns [`crate::Error`] when the database has no configured + /// [`crate::MergeOperator`]. + /// + /// # Panics + /// + /// Panics under the same input constraints as other write operations: + /// the key must be non-empty and no larger than `u16::MAX` bytes, and the + /// operand must be no larger than `u32::MAX` bytes. + /// + /// # Example + /// + /// ``` + /// use std::sync::Arc; + /// + /// use bytes::Bytes; + /// use slatedb::object_store::{memory::InMemory, ObjectStore}; + /// use slatedb::{Db, Error, IsolationLevel, MergeOperator, MergeOperatorError}; + /// + /// struct Counter; + /// + /// impl MergeOperator for Counter { + /// fn merge( + /// &self, + /// _key: &Bytes, + /// existing: Option, + /// operand: Bytes, + /// ) -> Result { + /// let current = existing + /// .map(|value| u64::from_le_bytes(value.as_ref().try_into().unwrap())) + /// .unwrap_or(0); + /// let delta = u64::from_le_bytes(operand.as_ref().try_into().unwrap()); + /// Ok(Bytes::copy_from_slice(&(current + delta).to_le_bytes())) + /// } + /// } + /// + /// # #[tokio::main] + /// # async fn main() -> Result<(), Error> { + /// let store: Arc = Arc::new(InMemory::new()); + /// let db = Db::builder("commutative-merge-example", store) + /// .with_merge_operator(Arc::new(Counter)) + /// .build() + /// .await?; + /// + /// let first = db.begin(IsolationLevel::SerializableSnapshot).await?; + /// let second = db.begin(IsolationLevel::SerializableSnapshot).await?; + /// first.merge_commutative(b"counter", 1_u64.to_le_bytes())?; + /// second.merge_commutative(b"counter", 1_u64.to_le_bytes())?; + /// first.commit().await?; + /// second.commit().await?; + /// + /// let value = db.get(b"counter").await?.unwrap(); + /// assert_eq!(u64::from_le_bytes(value.as_ref().try_into().unwrap()), 2); + /// # Ok(()) + /// # } + /// ``` + pub fn merge_commutative(&self, key: K, operand: V) -> Result<(), crate::Error> + where + K: AsRef<[u8]>, + V: AsRef<[u8]>, + { + if self.db_inner.flush_merge_operator.is_none() { + return Err(SlateDBError::MergeOperatorMissing.into()); + } + + self.write_batch.write().merge_commutative(key, operand); + Ok(()) + } + /// Merge a key-value pair into the transaction with custom options. /// /// ## Errors @@ -754,17 +871,25 @@ impl DbTransaction { return Ok(None); } - // Track only write keys that were not explicitly unmarked. - let tracked_write_keys = { + // Track only writes that were not explicitly unmarked. Compatibility + // is derived from the final surviving operations for each key. + let tracked_writes: HashMap = { let untracked_write_keys = self.untracked_write_keys.read(); write_batch .keys() .into_iter() .filter(|key| !untracked_write_keys.contains(key)) + .map(|key| { + let kind = if write_batch.is_commutative_merge_key(&key) { + TransactionWriteKind::CommutativeMerge + } else { + TransactionWriteKind::Exclusive + }; + (key, kind) + }) .collect() }; - self.txn_manager - .track_write_keys(&self.txn_id, &tracked_write_keys); + self.txn_manager.track_writes(&self.txn_id, &tracked_writes); // Submit the WriteBatch to the database for processing. The batch is sent to a // dedicated background task (in batch_write.rs) that processes all WriteBatches @@ -893,6 +1018,14 @@ impl DbTransactionOps for DbTransaction { DbTransaction::merge_with_options(self, key, value, options) } + fn merge_commutative(&self, key: K, operand: V) -> Result<(), crate::Error> + where + K: AsRef<[u8]>, + V: AsRef<[u8]>, + { + DbTransaction::merge_commutative(self, key, operand) + } + fn mark_read(&self, keys: I) -> Result<(), crate::Error> where K: AsRef<[u8]>, @@ -1162,6 +1295,64 @@ mod tests { assert!(txn1.commit().await.is_err()); } + #[tokio::test] + async fn test_txn_multi_get_reads_commutative_merge_overlay() { + let object_store: Arc = Arc::new(InMemory::new()); + let db = crate::Db::builder("test_txn_multi_get_commutative_overlay", object_store) + .with_merge_operator(Arc::new(CounterMergeOperator)) + .build() + .await + .unwrap(); + db.put(b"counter", 10_u64.to_le_bytes()).await.unwrap(); + + let txn = db.begin(IsolationLevel::Snapshot).await.unwrap(); + txn.merge_commutative(b"counter", 2_u64.to_le_bytes()) + .unwrap(); + + let values = txn + .multi_get(&[b"counter".as_ref(), b"counter".as_ref()]) + .await + .unwrap(); + for value in values { + let value = value.unwrap(); + assert_eq!(u64::from_le_bytes(value.as_ref().try_into().unwrap()), 12); + } + + txn.commit().await.unwrap(); + let value = db.get(b"counter").await.unwrap().unwrap(); + assert_eq!(u64::from_le_bytes(value.as_ref().try_into().unwrap()), 12); + } + + #[tokio::test] + async fn test_txn_multi_get_read_conflicts_with_commutative_merge() { + let object_store: Arc = Arc::new(InMemory::new()); + let db = crate::Db::builder("test_txn_multi_get_commutative_conflict", object_store) + .with_merge_operator(Arc::new(CounterMergeOperator)) + .build() + .await + .unwrap(); + db.put(b"counter", 0_u64.to_le_bytes()).await.unwrap(); + + let reader = db + .begin(IsolationLevel::SerializableSnapshot) + .await + .unwrap(); + let values = reader.multi_get(&[b"counter"]).await.unwrap(); + assert_eq!( + u64::from_le_bytes(values[0].as_ref().unwrap().as_ref().try_into().unwrap()), + 0 + ); + + let writer = db.begin(IsolationLevel::Snapshot).await.unwrap(); + writer + .merge_commutative(b"counter", 1_u64.to_le_bytes()) + .unwrap(); + writer.commit().await.unwrap(); + + reader.put(b"reader-write", b"value").unwrap(); + assert!(reader.commit().await.is_err()); + } + #[tokio::test] async fn test_txn_si_commit_conflict() { // Setup database with initial data @@ -2335,6 +2526,146 @@ mod tests { assert_eq!(total, EXPECTED); } + #[tokio::test(flavor = "multi_thread", worker_threads = 4)] + async fn test_commutative_merge_counter_aggregates_under_high_concurrency() { + const CONCURRENT_TXNS: usize = 32; + const ROUNDS: usize = 8; + const MERGE_INCREMENT: [u8; 8] = 1u64.to_le_bytes(); + const EXPECTED: u64 = (CONCURRENT_TXNS * ROUNDS) as u64; + + let object_store: Arc = Arc::new(InMemory::new()); + let db = crate::Db::builder("test_commutative_merge_counter", object_store) + .with_merge_operator(Arc::new(CounterMergeOperator)) + .build() + .await + .unwrap(); + + for _ in 0..ROUNDS { + let barrier = Arc::new(tokio::sync::Barrier::new(CONCURRENT_TXNS)); + let mut handles = Vec::with_capacity(CONCURRENT_TXNS); + + for _ in 0..CONCURRENT_TXNS { + let db = db.clone(); + let barrier = barrier.clone(); + handles.push(tokio::spawn(async move { + let txn = db + .begin(IsolationLevel::SerializableSnapshot) + .await + .unwrap(); + txn.merge_commutative(b"counter", MERGE_INCREMENT).unwrap(); + barrier.wait().await; + txn.commit().await.unwrap(); + })); + } + + for handle in handles { + handle.await.unwrap(); + } + } + + let value = db.get(b"counter").await.unwrap().unwrap(); + let total = u64::from_le_bytes(value.as_ref().try_into().unwrap()); + assert_eq!(total, EXPECTED); + } + + #[tokio::test] + async fn test_commutative_merge_conflicts_with_ordinary_merge_put_and_delete() { + const INITIAL: [u8; 8] = 0u64.to_le_bytes(); + const INCREMENT: [u8; 8] = 1u64.to_le_bytes(); + + let object_store: Arc = Arc::new(InMemory::new()); + let db = crate::Db::builder("test_commutative_merge_exclusive_conflicts", object_store) + .with_merge_operator(Arc::new(CounterMergeOperator)) + .build() + .await + .unwrap(); + + for key in [b"merge".as_slice(), b"put".as_slice(), b"delete".as_slice()] { + db.put(key, INITIAL).await.unwrap(); + let commutative = db.begin(IsolationLevel::Snapshot).await.unwrap(); + let exclusive = db.begin(IsolationLevel::Snapshot).await.unwrap(); + commutative.merge_commutative(key, INCREMENT).unwrap(); + + match key { + b"merge" => exclusive.merge(key, INCREMENT).unwrap(), + b"put" => exclusive.put(key, INITIAL).unwrap(), + b"delete" => exclusive.delete(key).unwrap(), + _ => unreachable!(), + } + + commutative.commit().await.unwrap(); + assert!(exclusive.commit().await.is_err()); + } + } + + #[tokio::test] + async fn test_mixed_same_key_operations_restore_exclusive_conflicts() { + const INITIAL: [u8; 8] = 0u64.to_le_bytes(); + const INCREMENT: [u8; 8] = 1u64.to_le_bytes(); + + let object_store: Arc = Arc::new(InMemory::new()); + let db = crate::Db::builder("test_commutative_merge_mixed_operations", object_store) + .with_merge_operator(Arc::new(CounterMergeOperator)) + .build() + .await + .unwrap(); + db.put(b"counter", INITIAL).await.unwrap(); + + let mixed = db.begin(IsolationLevel::Snapshot).await.unwrap(); + mixed.merge_commutative(b"counter", INCREMENT).unwrap(); + mixed.put(b"counter", INITIAL).unwrap(); + + let commutative = db.begin(IsolationLevel::Snapshot).await.unwrap(); + commutative + .merge_commutative(b"counter", INCREMENT) + .unwrap(); + commutative.commit().await.unwrap(); + + assert!(mixed.commit().await.is_err()); + } + + #[tokio::test] + async fn test_commutative_merge_remains_visible_to_ssi_point_and_range_reads() { + const INCREMENT: [u8; 8] = 1u64.to_le_bytes(); + + let object_store: Arc = Arc::new(InMemory::new()); + let db = crate::Db::builder("test_commutative_merge_ssi_reads", object_store) + .with_merge_operator(Arc::new(CounterMergeOperator)) + .build() + .await + .unwrap(); + + let point_reader = db + .begin(IsolationLevel::SerializableSnapshot) + .await + .unwrap(); + assert_eq!(point_reader.get(b"point").await.unwrap(), None); + let point_writer = db.begin(IsolationLevel::Snapshot).await.unwrap(); + point_writer.merge_commutative(b"point", INCREMENT).unwrap(); + point_writer.commit().await.unwrap(); + point_reader.put(b"point-reader-write", b"value").unwrap(); + assert!(point_reader.commit().await.is_err()); + + let range_reader = db + .begin(IsolationLevel::SerializableSnapshot) + .await + .unwrap(); + let mut range = range_reader + .scan(&b"range-a"[..]..=&b"range-z"[..]) + .await + .unwrap(); + while range.next().await.unwrap().is_some() {} + drop(range); + + let range_writer = db.begin(IsolationLevel::Snapshot).await.unwrap(); + range_writer + .merge_commutative(b"range-m", INCREMENT) + .unwrap(); + range_writer.commit().await.unwrap(); + range_reader.put(b"range-reader-write", b"value").unwrap(); + assert!(range_reader.commit().await.is_err()); + } + #[tokio::test] async fn test_txn_merge_requires_merge_operator() { let object_store: Arc = Arc::new(InMemory::new()); @@ -2352,6 +2683,20 @@ mod tests { assert_eq!(db.get(b"counter").await.unwrap(), None); } + #[tokio::test] + async fn test_txn_commutative_merge_requires_merge_operator() { + let object_store: Arc = Arc::new(InMemory::new()); + let db = crate::Db::open("test_txn_commutative_merge_requires_operator", object_store) + .await + .unwrap(); + + let txn = db.begin(IsolationLevel::Snapshot).await.unwrap(); + let err = txn + .merge_commutative(b"counter", 1u64.to_le_bytes()) + .unwrap_err(); + assert_eq!(err.kind(), crate::ErrorKind::Invalid); + } + #[tokio::test] async fn test_txn_commit_rejects_same_key_merge_different_ttls() { let object_store: Arc = Arc::new(InMemory::new()); diff --git a/slatedb/src/ops.rs b/slatedb/src/ops.rs index 87a31a62b0..4a9bb6d6ac 100644 --- a/slatedb/src/ops.rs +++ b/slatedb/src/ops.rs @@ -519,6 +519,25 @@ pub trait DbTransactionOps: DbReadOps { self.merge_with_options(key, value, &MergeOptions::default()) } + /// Buffers a merge operand that is compatible with concurrent + /// commutative merges for the same key. + /// + /// See [`DbTransaction::merge_commutative`](crate::DbTransaction::merge_commutative) + /// for the algebraic caller contract and the precise conflict, isolation, + /// atomicity, and persistence guarantees. + /// + /// The default implementation conservatively delegates to [`Self::merge`] + /// so third-party transaction implementations remain source-compatible and + /// retain ordinary write/write conflicts unless they explicitly override + /// this method. + fn merge_commutative(&self, key: K, operand: V) -> Result<(), crate::Error> + where + K: AsRef<[u8]>, + V: AsRef<[u8]>, + { + self.merge(key, operand) + } + /// Merge a key-value pair into the transaction with custom `MergeOptions`. /// /// ## Errors diff --git a/slatedb/src/transaction_manager.rs b/slatedb/src/transaction_manager.rs index e6df1cb077..70ef591ae0 100644 --- a/slatedb/src/transaction_manager.rs +++ b/slatedb/src/transaction_manager.rs @@ -21,6 +21,16 @@ pub enum IsolationLevel { SerializableSnapshot, } +/// Conflict behavior for a transaction's final operations on one key. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub(crate) enum TransactionWriteKind { + /// The key conflicts with every concurrent write to the same key. + Exclusive, + /// The key is merge-only and may coexist with another merge-only, + /// commutative write to the same key. + CommutativeMerge, +} + #[derive(Debug)] pub(crate) struct TransactionState { /// The sequence number when the transaction started. This is used to establish @@ -30,9 +40,9 @@ pub(crate) struct TransactionState { /// a transaction is committed and is used to check conflicts with recent committed /// transactions. committed_seq: Option, - /// The write keys of the transaction for write-write conflict detection. + /// The transaction's final per-key write behavior for conflict detection. /// Used in both Snapshot Isolation and Serializable Snapshot Isolation. - write_keys: HashSet, + writes: HashMap, /// The read keys of the transaction for read-write conflict detection. /// Only used in Serializable Snapshot Isolation mode. read_keys: HashSet, @@ -42,9 +52,29 @@ pub(crate) struct TransactionState { } impl TransactionState { - /// Add write keys to this transaction's write set for conflict detection. + /// Add writes to this transaction's write set for conflict detection. + /// + /// Exclusive behavior dominates when a key is tracked more than once so a + /// mixed operation sequence cannot accidentally become compatible. + fn track_writes(&mut self, writes: impl IntoIterator) { + for (key, kind) in writes { + self.writes + .entry(key) + .and_modify(|existing| { + if *existing != kind { + *existing = TransactionWriteKind::Exclusive; + } + }) + .or_insert(kind); + } + } + + #[cfg(test)] fn track_write_keys(&mut self, keys: impl IntoIterator) { - self.write_keys.extend(keys); + self.track_writes( + keys.into_iter() + .map(|key| (key, TransactionWriteKind::Exclusive)), + ); } /// Add read keys to this transaction's read set for SSI conflict detection. @@ -120,7 +150,7 @@ impl TransactionManager { TransactionState { started_seq: seq, committed_seq: None, - write_keys: HashSet::new(), + writes: HashMap::new(), read_keys: HashSet::new(), read_ranges: Vec::new(), }, @@ -134,7 +164,7 @@ impl TransactionManager { let txn_state = TransactionState { started_seq: seq, committed_seq: None, - write_keys: HashSet::new(), + writes: HashMap::new(), read_keys: HashSet::new(), read_ranges: Vec::new(), }; @@ -158,15 +188,29 @@ impl TransactionManager { inner.recycle_recent_committed_txns(); } - /// Track write keys for a transaction. This is used for conflict detection. - /// Keys should be tracked before calling commit-related methods. - pub(crate) fn track_write_keys(&self, txn_id: &Uuid, write_keys: &HashSet) { + /// Track writes for a transaction. This is used for conflict detection. + /// Writes should be tracked before calling commit-related methods. + pub(crate) fn track_writes( + &self, + txn_id: &Uuid, + writes: &HashMap, + ) { let mut inner = self.inner.write(); if let Some(txn_state) = inner.active_txns.get_mut(txn_id) { - txn_state.track_write_keys(write_keys.iter().cloned()); + txn_state.track_writes(writes.iter().map(|(key, kind)| (key.clone(), *kind))); } } + #[cfg(test)] + fn track_write_keys(&self, txn_id: &Uuid, write_keys: &HashSet) { + let writes = write_keys + .iter() + .cloned() + .map(|key| (key, TransactionWriteKind::Exclusive)) + .collect(); + self.track_writes(txn_id, &writes); + } + /// Track a key read operation (for SSI) pub(crate) fn track_read_keys( &self, @@ -200,8 +244,7 @@ impl TransactionManager { }; // both SI and SSI need to check write-write conflicts - let ww_conflict = - inner.has_write_write_conflict(&txn_state.write_keys, txn_state.started_seq); + let ww_conflict = inner.has_write_write_conflict(&txn_state.writes, txn_state.started_seq); if ww_conflict { return true; } @@ -249,7 +292,11 @@ impl TransactionManager { inner.track_recent_committed_state(TransactionState { started_seq: committed_seq, committed_seq: Some(committed_seq), - write_keys: keys.clone(), + writes: keys + .iter() + .cloned() + .map(|key| (key, TransactionWriteKind::Exclusive)) + .collect(), read_keys: HashSet::new(), read_ranges: Vec::new(), }); @@ -319,9 +366,13 @@ impl TransactionManagerInner { } } - fn has_write_write_conflict(&self, write_keys: &HashSet, started_seq: u64) -> bool { + fn has_write_write_conflict( + &self, + writes: &HashMap, + started_seq: u64, + ) -> bool { // If the current transaction has no write operations, there's no write-write conflict - if write_keys.is_empty() { + if writes.is_empty() { return false; } @@ -331,12 +382,21 @@ impl TransactionManagerInner { "all txns in recent_committed_txns should be committed with committed_seq set", ); - // if another transaction committed after the current transaction started, - // and they have overlapping write keys, then there's a conflict. - if other_committed_seq > started_seq - && !write_keys.is_disjoint(&committed_txn.write_keys) - { - return true; + if other_committed_seq > started_seq { + for (key, kind) in writes { + let Some(other_kind) = committed_txn.writes.get(key) else { + continue; + }; + if !matches!( + (kind, other_kind), + ( + TransactionWriteKind::CommutativeMerge, + TransactionWriteKind::CommutativeMerge + ) + ) { + return true; + } + } } } @@ -368,7 +428,10 @@ impl TransactionManagerInner { if other_committed_seq > started_seq { // Check if any of the current transaction's read keys were written by // the committed transaction. - if !read_keys.is_disjoint(&committed_txn.write_keys) { + if read_keys + .iter() + .any(|read_key| committed_txn.writes.contains_key(read_key)) + { return true; } @@ -376,8 +439,8 @@ impl TransactionManagerInner { // committed transaction write keys. for read_range in &read_ranges { if committed_txn - .write_keys - .iter() + .writes + .keys() .any(|write_key| read_range.contains(write_key)) { return true; @@ -400,6 +463,22 @@ mod tests { use slatedb_common::DbRand; use std::collections::HashSet; + fn exclusive_writes( + keys: impl IntoIterator, + ) -> HashMap { + keys.into_iter() + .map(|key| (Bytes::from(key), TransactionWriteKind::Exclusive)) + .collect() + } + + fn commutative_writes( + keys: impl IntoIterator, + ) -> HashMap { + keys.into_iter() + .map(|key| (Bytes::from(key), TransactionWriteKind::CommutativeMerge)) + .collect() + } + struct CheckConflictTestCase { name: &'static str, recent_committed_txns: Vec, @@ -495,7 +574,7 @@ mod tests { recent_committed_txns: vec![TransactionState { started_seq: 50, committed_seq: Some(80), - write_keys: ["key1", "key2"].into_iter().map(Bytes::from).collect(), + writes: exclusive_writes(["key1", "key2"]), read_keys: HashSet::new(), read_ranges: Vec::new(), }], @@ -508,7 +587,7 @@ mod tests { recent_committed_txns: vec![TransactionState { started_seq: 50, committed_seq: Some(150), - write_keys: ["key1"].into_iter().map(Bytes::from).collect(), + writes: exclusive_writes(["key1"]), read_keys: HashSet::new(), read_ranges: Vec::new(), }], @@ -522,14 +601,14 @@ mod tests { TransactionState { started_seq: 30, committed_seq: Some(50), - write_keys: ["key1"].into_iter().map(Bytes::from).collect(), + writes: exclusive_writes(["key1"]), read_keys: HashSet::new(), read_ranges: Vec::new(), }, TransactionState { started_seq: 80, committed_seq: Some(150), - write_keys: ["key2"].into_iter().map(Bytes::from).collect(), + writes: exclusive_writes(["key2"]), read_keys: HashSet::new(), read_ranges: Vec::new(), }, @@ -543,7 +622,7 @@ mod tests { recent_committed_txns: vec![TransactionState { started_seq: 30, committed_seq: Some(50), - write_keys: ["key1"].into_iter().map(Bytes::from).collect(), + writes: exclusive_writes(["key1"]), read_keys: HashSet::new(), read_ranges: Vec::new(), }], @@ -556,7 +635,7 @@ mod tests { recent_committed_txns: vec![TransactionState { started_seq: 100, committed_seq: Some(100), - write_keys: ["key1"].into_iter().map(Bytes::from).collect(), + writes: exclusive_writes(["key1"]), read_keys: HashSet::new(), read_ranges: Vec::new(), }], @@ -569,10 +648,7 @@ mod tests { recent_committed_txns: vec![TransactionState { started_seq: 80, committed_seq: Some(150), - write_keys: ["key1", "key2", "key3"] - .into_iter() - .map(Bytes::from) - .collect(), + writes: exclusive_writes(["key1", "key2", "key3"]), read_keys: HashSet::new(), read_ranges: Vec::new(), }], @@ -585,7 +661,7 @@ mod tests { recent_committed_txns: vec![TransactionState { started_seq: u64::MAX - 1, committed_seq: Some(u64::MAX), - write_keys: ["key1"].into_iter().map(Bytes::from).collect(), + writes: exclusive_writes(["key1"]), read_keys: HashSet::new(), read_ranges: Vec::new(), }], @@ -603,15 +679,12 @@ mod tests { } // Convert current transaction write keys - let conflict_keys: HashSet = case - .current_write_keys - .into_iter() - .map(Bytes::from) - .collect(); + let conflict_writes = exclusive_writes(case.current_write_keys); // Call the method under test let inner = txn_manager.inner.read(); - let has_conflict = inner.has_write_write_conflict(&conflict_keys, case.current_started_seq); + let has_conflict = + inner.has_write_write_conflict(&conflict_writes, case.current_started_seq); // Verify result assert_eq!( @@ -621,6 +694,131 @@ mod tests { ); } + #[rstest] + #[case::commutative_with_commutative( + TransactionWriteKind::CommutativeMerge, + TransactionWriteKind::CommutativeMerge, + false + )] + #[case::commutative_with_exclusive( + TransactionWriteKind::CommutativeMerge, + TransactionWriteKind::Exclusive, + true + )] + #[case::exclusive_with_commutative( + TransactionWriteKind::Exclusive, + TransactionWriteKind::CommutativeMerge, + true + )] + #[case::exclusive_with_exclusive( + TransactionWriteKind::Exclusive, + TransactionWriteKind::Exclusive, + true + )] + fn test_write_write_conflict_respects_write_kind( + #[case] current_kind: TransactionWriteKind, + #[case] committed_kind: TransactionWriteKind, + #[case] expected_conflict: bool, + ) { + let txn_manager = create_transaction_manager(); + txn_manager + .inner + .write() + .recent_committed_txns + .push_back(TransactionState { + started_seq: 50, + committed_seq: Some(150), + writes: [(Bytes::from_static(b"key"), committed_kind)] + .into_iter() + .collect(), + read_keys: HashSet::new(), + read_ranges: Vec::new(), + }); + let writes = [(Bytes::from_static(b"key"), current_kind)] + .into_iter() + .collect(); + + assert_eq!( + txn_manager + .inner + .read() + .has_write_write_conflict(&writes, 100), + expected_conflict + ); + } + + #[test] + fn test_mixed_write_kinds_are_exclusive_in_either_order() { + let mut repeated_commutative = TransactionState { + started_seq: 0, + committed_seq: None, + writes: HashMap::new(), + read_keys: HashSet::new(), + read_ranges: Vec::new(), + }; + repeated_commutative.track_writes(commutative_writes(["key"])); + repeated_commutative.track_writes(commutative_writes(["key"])); + assert_eq!( + repeated_commutative.writes.get(b"key".as_slice()), + Some(&TransactionWriteKind::CommutativeMerge) + ); + + let mut commutative_then_exclusive = TransactionState { + started_seq: 0, + committed_seq: None, + writes: HashMap::new(), + read_keys: HashSet::new(), + read_ranges: Vec::new(), + }; + commutative_then_exclusive.track_writes(commutative_writes(["key"])); + commutative_then_exclusive.track_writes(exclusive_writes(["key"])); + assert_eq!( + commutative_then_exclusive.writes.get(b"key".as_slice()), + Some(&TransactionWriteKind::Exclusive) + ); + + let mut exclusive_then_commutative = TransactionState { + started_seq: 0, + committed_seq: None, + writes: HashMap::new(), + read_keys: HashSet::new(), + read_ranges: Vec::new(), + }; + exclusive_then_commutative.track_writes(exclusive_writes(["key"])); + exclusive_then_commutative.track_writes(commutative_writes(["key"])); + assert_eq!( + exclusive_then_commutative.writes.get(b"key".as_slice()), + Some(&TransactionWriteKind::Exclusive) + ); + } + + #[test] + fn test_commutative_writes_remain_visible_to_point_and_range_read_conflicts() { + let txn_manager = create_transaction_manager(); + txn_manager + .inner + .write() + .recent_committed_txns + .push_back(TransactionState { + started_seq: 50, + committed_seq: Some(150), + writes: commutative_writes(["foo5"]), + read_keys: HashSet::new(), + read_ranges: Vec::new(), + }); + let read_keys = [Bytes::from_static(b"foo5")].into_iter().collect(); + + let inner = txn_manager.inner.read(); + assert!(inner.has_read_write_conflict(&read_keys, Vec::new(), 100)); + assert!(inner.has_read_write_conflict( + &HashSet::new(), + vec![BytesRange::from( + Bytes::from_static(b"foo0")..=Bytes::from_static(b"foo9") + )], + 100 + )); + } + #[derive(Debug)] struct MinActiveSeqTestCase { name: &'static str, @@ -686,7 +884,7 @@ mod tests { expected_recent_committed_txn: Some(TransactionState { started_seq: 100, committed_seq: Some(150), - write_keys: ["key1", "key2"].into_iter().map(Bytes::from).collect(), + writes: exclusive_writes(["key1", "key2"]), read_keys: HashSet::new(), read_ranges: Vec::new(), }), @@ -720,7 +918,7 @@ mod tests { expected_recent_committed_txn: Some(TransactionState { started_seq: 100, committed_seq: Some(100), - write_keys: ["key1", "key2"].into_iter().map(Bytes::from).collect(), + writes: exclusive_writes(["key1", "key2"]), read_keys: HashSet::new(), read_ranges: Vec::new(), }), @@ -751,7 +949,7 @@ mod tests { expected_recent_committed_txn: Some(TransactionState { started_seq: 100, committed_seq: Some(150), - write_keys: ["existing_key", "key1", "key2"].into_iter().map(Bytes::from).collect(), + writes: exclusive_writes(["existing_key", "key1", "key2"]), read_keys: HashSet::new(), read_ranges: Vec::new(), }), @@ -806,9 +1004,9 @@ mod tests { test_case.name, expected_txn.committed_seq, actual_txn.committed_seq ); assert_eq!( - actual_txn.write_keys, expected_txn.write_keys, - "Test case '{}' failed: expected write_keys {:?}, got {:?}", - test_case.name, expected_txn.write_keys, actual_txn.write_keys + actual_txn.writes, expected_txn.writes, + "Test case '{}' failed: expected writes {:?}, got {:?}", + test_case.name, expected_txn.writes, actual_txn.writes ); } } @@ -919,7 +1117,7 @@ mod tests { inner.recent_committed_txns.push_back(TransactionState { started_seq: 50, committed_seq: None, // This should not happen in practice but let's test - write_keys: HashSet::new(), + writes: HashMap::new(), read_keys: HashSet::new(), read_ranges: Vec::new(), }); @@ -1403,13 +1601,16 @@ mod tests { } // Direct read-write conflict on keys - let direct_conflict = !txn.read_keys.is_disjoint(&committed.write_keys); + let direct_conflict = txn + .read_keys + .iter() + .any(|read_key| committed.writes.contains_key(read_key)); // Phantom conflict via range containment let mut phantom_conflict = false; - if !txn.read_ranges.is_empty() && !committed.write_keys.is_empty() { + if !txn.read_ranges.is_empty() && !committed.writes.is_empty() { 'outer: for range in txn.read_ranges.iter() { - for w in committed.write_keys.iter() { + for w in committed.writes.keys() { if range.contains(w) { phantom_conflict = true; break 'outer; From 822c97b2ff0ea41d6e59c325990ae5b80b70e91c Mon Sep 17 00:00:00 2001 From: Matthew Sanetra <41018997+matthewsanetra@users.noreply.github.com> Date: Fri, 14 Aug 2026 16:01:50 +0100 Subject: [PATCH 62/65] Restore HelixDB cache and reader compatibility Keep the fork-facing object-store cache switch while routing it through v0.15's split flush and compaction admission policy. Accept the legacy optional checkpoint argument at DbReader::open through DbReaderMode conversion without weakening the v0.15 builder API. --- .../src/cached_object_store/object_store.rs | 79 +++++++++++++++++-- slatedb/src/config.rs | 17 ++-- slatedb/src/db_reader.rs | 23 +++++- 3 files changed, 97 insertions(+), 22 deletions(-) diff --git a/slatedb/src/cached_object_store/object_store.rs b/slatedb/src/cached_object_store/object_store.rs index 47dbaf1f3f..d7e6d9a447 100644 --- a/slatedb/src/cached_object_store/object_store.rs +++ b/slatedb/src/cached_object_store/object_store.rs @@ -138,6 +138,28 @@ impl CachedObjectStore { recorder: &MetricsRecorderHelper, clock: Arc, rand: Arc, + ) -> Result>, SlateDBError> { + Self::from_config_with_cache_put_config( + object_store, + options, + CachePutConfig { + cache_on_flush: options.cache_puts, + cache_on_compaction: options.cache_puts, + }, + recorder, + clock, + rand, + ) + .await + } + + async fn from_config_with_cache_put_config( + object_store: Arc, + options: &ObjectStoreCacheOptions, + cache_put_config: CachePutConfig, + recorder: &MetricsRecorderHelper, + clock: Arc, + rand: Arc, ) -> Result>, SlateDBError> { let cache_root_folder = match &options.root_folder { None => return Ok(None), @@ -157,10 +179,7 @@ impl CachedObjectStore { object_store, cache_storage, options.part_size_bytes, - CachePutConfig { - cache_on_flush: options.cache_on_flush, - cache_on_compaction: options.cache_on_compaction, - }, + cache_put_config, stats, )?; cached.start_evictor().await; @@ -179,6 +198,7 @@ impl CachedObjectStore { root_folder: Some(root_folder.into()), ..ObjectStoreCacheOptions::default() }, + cache_put_config: CachePutConfig::default(), metrics_recorder: Arc::new(NoopMetricsRecorder::new()), metric_level: MetricLevel::default(), } @@ -731,6 +751,7 @@ impl CachedObjectStore { pub struct CachedObjectStoreBuilder { object_store: Arc, options: ObjectStoreCacheOptions, + cache_put_config: CachePutConfig, metrics_recorder: Arc, metric_level: MetricLevel, } @@ -757,7 +778,7 @@ impl CachedObjectStoreBuilder { /// /// The default is false. pub fn with_cache_on_flush(mut self, cache_on_flush: bool) -> Self { - self.options.cache_on_flush = cache_on_flush; + self.cache_put_config.cache_on_flush = cache_on_flush; self } @@ -766,7 +787,7 @@ impl CachedObjectStoreBuilder { /// /// The default is false. pub fn with_cache_on_compaction(mut self, cache_on_compaction: bool) -> Self { - self.options.cache_on_compaction = cache_on_compaction; + self.cache_put_config.cache_on_compaction = cache_on_compaction; self } @@ -808,9 +829,10 @@ impl CachedObjectStoreBuilder { /// Builds the `CachedObjectStore` and starts its evictor. pub async fn build(self) -> Result, crate::Error> { let recorder = MetricsRecorderHelper::new(self.metrics_recorder, self.metric_level); - let cached = CachedObjectStore::from_config( + let cached = CachedObjectStore::from_config_with_cache_put_config( self.object_store, &self.options, + self.cache_put_config, &recorder, Arc::new(DefaultSystemClock::new()), Arc::new(DbRand::default()), @@ -1187,6 +1209,7 @@ mod tests { use crate::cached_object_store::storage::{LocalCacheStorage, PartID}; use crate::cached_object_store::storage_fs::FsCacheEntry; use crate::cached_object_store::storage_fs::FsCacheStorage; + use crate::config::ObjectStoreCacheOptions; use crate::db_state::SstType; use crate::instrumented_object_store::{InstrumentedObjectStore, ObjectStoreComponent}; use crate::object_store_tag::{ObjectStoreCallTag, TableStoreKind}; @@ -2509,6 +2532,48 @@ mod tests { assert_eq!(cached_part_count(&store, &location).await, expected_parts); } + #[tokio::test] + async fn test_cache_puts_config_admits_flush_and_compaction_writes() { + let upstream: Arc = Arc::new(object_store::memory::InMemory::new()); + let options = ObjectStoreCacheOptions { + root_folder: Some(new_test_cache_folder()), + part_size_bytes: 1024, + cache_puts: true, + ..ObjectStoreCacheOptions::default() + }; + let store = CachedObjectStore::from_config( + upstream, + &options, + &MetricsRecorderHelper::noop(), + Arc::new(DefaultSystemClock::new()), + Arc::new(DbRand::default()), + ) + .await + .unwrap() + .unwrap(); + + for (location, tag) in [ + ( + Path::from("compacted/flush.sst"), + ObjectStoreCallTag::new(TableStoreKind::Main, SstType::Compacted), + ), + ( + Path::from("compacted/compaction.sst"), + ObjectStoreCallTag::new(TableStoreKind::Compactor, SstType::Compacted), + ), + ] { + store + .put_opts( + &location, + PutPayload::from_bytes(gen_rand_bytes(2048)), + put_opts_tagged(tag), + ) + .await + .unwrap(); + assert_eq!(cached_part_count(&store, &location).await, 2); + } + } + #[tokio::test] async fn test_untagged_put_is_not_cached() { let upstream: Arc = Arc::new(object_store::memory::InMemory::new()); diff --git a/slatedb/src/config.rs b/slatedb/src/config.rs index bfe418d9cb..f9f8b8dffc 100644 --- a/slatedb/src/config.rs +++ b/slatedb/src/config.rs @@ -1726,17 +1726,11 @@ pub struct ObjectStoreCacheOptions { /// its default value is 4mb. pub part_size_bytes: usize, - /// Whether to cache compacted SSTs produced by memtable flushes to the - /// local disk cache, for faster subsequent reads. + /// Whether to cache compacted SSTs produced by memtable flushes or + /// compaction to the local disk cache, for faster subsequent reads. /// - /// Default is false. - pub cache_on_flush: bool, - - /// Whether to cache compacted SSTs produced by compaction to the local - /// disk cache, for faster subsequent reads. - /// - /// Default is false. - pub cache_on_compaction: bool, + /// WAL writes are never cached. Default is false. + pub cache_puts: bool, /// Whether to preload SST files into cache during database startup. When enabled, /// the database will load SST files into the cache up to the cache size limit @@ -1768,8 +1762,7 @@ impl Default for ObjectStoreCacheOptions { #[cfg(not(target_pointer_width = "32"))] max_cache_size_bytes: Some(16 * 1024 * 1024 * 1024), part_size_bytes: 4 * 1024 * 1024, - cache_on_flush: false, - cache_on_compaction: false, + cache_puts: false, preload_disk_cache_on_startup: None, scan_interval: Some(Duration::from_secs(3600)), max_open_file_handles: 1000, diff --git a/slatedb/src/db_reader.rs b/slatedb/src/db_reader.rs index 6ab8e7e25e..0dccc5feed 100644 --- a/slatedb/src/db_reader.rs +++ b/slatedb/src/db_reader.rs @@ -78,6 +78,12 @@ pub enum DbReaderMode { FollowLatest, } +impl From> for DbReaderMode { + fn from(checkpoint: Option) -> Self { + checkpoint.map_or(Self::ManagedCheckpoint, Self::Checkpoint) + } +} + /// Where a reader stops replaying the WAL when it builds its state. /// /// This is only reached when replay is wanted at all; a reader configured with @@ -1386,16 +1392,16 @@ impl DbReader { /// pinned to a supplied checkpoint, or follows the latest manifest without GC protection. /// Managed readers retain each generation's checkpoint until all snapshots, iterators, and /// in-flight reads using that generation are gone. - pub async fn open>( + pub async fn open, M: Into>( path: P, object_store: Arc, - mode: DbReaderMode, + mode: M, options: DbReaderOptions, ) -> Result { // Use the builder API internally Self::builder(path, object_store) .with_options(options) - .with_reader_mode(mode) + .with_reader_mode(mode.into()) .build() .await } @@ -2157,6 +2163,17 @@ mod tests { } } + #[test] + fn legacy_checkpoint_options_convert_to_reader_modes() { + let checkpoint_id = Uuid::new_v4(); + + assert_eq!(DbReaderMode::from(None), DbReaderMode::ManagedCheckpoint); + assert_eq!( + DbReaderMode::from(Some(checkpoint_id)), + DbReaderMode::Checkpoint(checkpoint_id) + ); + } + async fn wait_for_reader_generation_change(reader: &DbReader, previous: Uuid) { tokio::time::timeout(Duration::from_secs(5), async { loop { From b418694a4fccac90b7181bbe5f4fb93f96586e9b Mon Sep 17 00:00:00 2001 From: Matthew Sanetra <41018997+matthewsanetra@users.noreply.github.com> Date: Tue, 1 Sep 2026 15:12:28 +0100 Subject: [PATCH 63/65] Adapt fork changes to SlateDB v0.16 --- Cargo.lock | 907 ++++++++---------- slatedb/benches/db_reader_memory_scaling.rs | 5 +- slatedb/benches/db_reader_scaling.rs | 20 +- slatedb/src/cached_object_store/storage_fs.rs | 4 +- slatedb/src/db_reader.rs | 43 +- slatedb/src/db_transaction.rs | 2 +- slatedb/src/error.rs | 4 +- 7 files changed, 415 insertions(+), 570 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index 18e335f2e3..1677f23ded 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -19,9 +19,9 @@ checksum = "320119579fcad9c21884f5c4861d16174d0e06250625266f50fe6898340abefa" [[package]] name = "aho-corasick" -version = "1.1.4" +version = "1.1.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ddd31a130427c27518df266943a5308ed92d4b226cc639f5a8f1002816174301" +checksum = "c982642fa9e8606056828ee9a8505737230110bb1099153c79efe865c59d12ba" dependencies = [ "memchr", ] @@ -114,15 +114,15 @@ dependencies = [ [[package]] name = "anyhow" -version = "1.0.102" +version = "1.0.104" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7f202df86484c868dbad7eaa557ef785d5c66295e41b460ef922eca0723b842c" +checksum = "330a5ed07fa54e4702c9d6c4174f74427fc0ef6e214bbd677ae50a5099946470" [[package]] name = "arrayvec" -version = "0.7.6" +version = "0.7.8" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7c02d123df017efcdfbd739ef81735b36c5ba83ec3c59c80a9d7ecc718f92e50" +checksum = "d3fb67a6e08acf24fdeccbac2cb6ac4305825bd1f117462e0e6f2f193345ad56" [[package]] name = "askama" @@ -151,7 +151,7 @@ dependencies = [ "rustc-hash", "serde", "serde_derive", - "syn", + "syn 2.0.119", ] [[package]] @@ -204,13 +204,13 @@ dependencies = [ [[package]] name = "async-trait" -version = "0.1.89" +version = "0.1.91" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9035ad2d096bed7955a320ee7e2230574d28fd3c3a0f186cbea1ff3c7eed5dbb" +checksum = "ae36dc4177970ef04fde5178d3e2429882def40e57a451f919c098f72baa6cec" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 3.0.4", ] [[package]] @@ -230,15 +230,15 @@ checksum = "1505bd5d3d116872e7271a6d4e16d81d0c8570876c8de68093a09ac269d8aac0" [[package]] name = "autocfg" -version = "1.5.0" +version = "1.5.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c08606f8c3cbf4ce6ec8e28fb0014a2c086708fe954eaa885384a6165172e7e8" +checksum = "f2032f911046de80f0a198e0901378627c33f59ea0ac00e363d481118bd70a53" [[package]] name = "aws-lc-rs" -version = "1.17.0" +version = "1.17.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5ec2f1fc3ec205783a5da9a7e6c1509cc69dedf09a1949e412c1e18469326d00" +checksum = "00bdb5da18dac48ca2cc7cd4a98e533e8635a58e2361d13a1a4ee3888e0d72f1" dependencies = [ "aws-lc-sys", "zeroize", @@ -246,14 +246,15 @@ dependencies = [ [[package]] name = "aws-lc-sys" -version = "0.41.0" +version = "0.43.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1a2f9779ce85b93ab6170dd940ad0169b5766ff848247aff13bb788b832fe3f4" +checksum = "43103168cc76fe62678a375e722fc9cb3a0146159ac5828bc4f0dfd755c2224c" dependencies = [ "cc", "cmake", "dunce", "fs_extra", + "pkg-config", ] [[package]] @@ -329,9 +330,9 @@ checksum = "bef38d45163c2f1dde094a7dfd33ccf595c92905c8f8f4fdc18d06fb1037718a" [[package]] name = "bitflags" -version = "2.11.1" +version = "2.13.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c4512299f36f043ab09a583e57bceb5a5aab7a73db1805848e8fef3c9e8c78b3" +checksum = "b588b76d00fde79687d7646a9b5bdf3cc0f655e0bbd080335a95d7e96f3587da" [[package]] name = "block-buffer" @@ -344,30 +345,30 @@ dependencies = [ [[package]] name = "bumpalo" -version = "3.20.2" +version = "3.20.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5d20789868f4b01b2f2caec9f5c4e0213b41e3e5702a50157d699ae31ced2fcb" +checksum = "72f5acc6cb2ba439de613abc23857ec3d78374d8ed5ac84e9d11336e87da8649" [[package]] name = "bytemuck" -version = "1.25.0" +version = "1.25.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c8efb64bd706a16a1bdde310ae86b351e4d21550d98d056f22f8a7f7a2183fec" +checksum = "95832e849adfb21180ccb6826a99da14e5d266ae5c2e668e1602cf234f153797" [[package]] name = "bytes" -version = "1.11.1" +version = "1.12.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1e748733b7cbc798e1434b6ac524f0c1ff2ab456fe201501e6497c8417a4fc33" +checksum = "fc652a48c352aef3ea3aed32080501cf3ef6ed5da78602a020c991775b0aff04" dependencies = [ "serde", ] [[package]] name = "camino" -version = "1.2.2" +version = "1.2.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e629a66d692cb9ff1a1c664e41771b3dcaf961985a9774c0eb0bd1b51cf60a48" +checksum = "bb1307f12aa967b5a58416e87b3653360e0fd614a016b6e970db08fecbb1b80d" dependencies = [ "serde_core", ] @@ -392,7 +393,7 @@ dependencies = [ "semver", "serde", "serde_json", - "thiserror 2.0.18", + "thiserror 2.0.20", ] [[package]] @@ -403,9 +404,9 @@ checksum = "37b2a672a2cb129a2e41c10b1224bb368f9f37a2b16b612598138befd7b37eb5" [[package]] name = "cc" -version = "1.2.62" +version = "1.4.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a1dce859f0832a7d088c4f1119888ab94ef4b5d6795d1ce05afb7fe159d79f98" +checksum = "5add81bb678e6cb321aff7fa0dc7689ad82b112dbc032cea19f91d6b8e3582b9" dependencies = [ "find-msvc-tools", "jobserver", @@ -421,15 +422,15 @@ checksum = "9330f8b2ff13f34540b44e946ef35111825727b38d33286ef986142615121801" [[package]] name = "cfg_aliases" -version = "0.2.1" +version = "0.2.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "613afe47fcd5fac7ccf1db93babcb082c5994d996f20b8b159f2ad1658eb5724" +checksum = "f079e83a288787bcd14a6aea84cee5c87a67c5a3e660c30f557a3d24761b3527" [[package]] name = "chacha20" -version = "0.10.0" +version = "0.10.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6f8d983286843e49675a4b7a2d174efe136dc93a18d69130dd18198a6c167601" +checksum = "d524456ba66e72eb8b115ff89e01e497f8e6d11d78b70b1aa13c0fbd97540a81" dependencies = [ "cfg-if", "cpufeatures", @@ -438,9 +439,9 @@ dependencies = [ [[package]] name = "chrono" -version = "0.4.44" +version = "0.4.45" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c673075a2e0e5f4a1dde27ce9dee1ea4558c7ffe648f576438a20ca1d2acc4b0" +checksum = "1aa79e62e7697b8e29b513a68abacf485adcd1fe8284a4316c5ae868e6633327" dependencies = [ "iana-time-zone", "js-sys", @@ -479,9 +480,9 @@ dependencies = [ [[package]] name = "clap" -version = "4.6.1" +version = "4.6.6" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1ddb117e43bbf7dacf0a4190fef4d345b9bad68dfc649cb349e7d17d28428e51" +checksum = "473c7e07f409a8d772161724aa8db6a765a2532a70f9667eeb7b49d3d02fbdca" dependencies = [ "clap_builder", "clap_derive", @@ -489,9 +490,9 @@ dependencies = [ [[package]] name = "clap_builder" -version = "4.6.0" +version = "4.6.6" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "714a53001bf66416adb0e2ef5ac857140e7dc3a0c48fb28b2f10762fc4b5069f" +checksum = "7b48fea5a88e9ae728a2dcbedbfc0e730f7d60da42e1cb049a83c9fb8b789889" dependencies = [ "anstream", "anstyle", @@ -501,14 +502,14 @@ dependencies = [ [[package]] name = "clap_derive" -version = "4.6.1" +version = "4.6.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f2ce8604710f6733aa641a2b3731eaa1e8b3d9973d5e3565da11800813f997a9" +checksum = "d012d2b9d65aca7f18f4d9878a045bc17899bba951561ba5ec3c2ba1eed9a061" dependencies = [ "heck 0.5.0", "proc-macro2", "quote", - "syn", + "syn 3.0.4", ] [[package]] @@ -559,9 +560,9 @@ dependencies = [ [[package]] name = "console" -version = "0.16.3" +version = "0.16.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d64e8af5551369d19cf50138de61f1c42074ab970f74e99be916646777f8fc87" +checksum = "4fe5f465a4f6fee88fad41b85d990f84c835335e85b5d9e6e63e0d06d28cba7c" dependencies = [ "encode_unicode", "libc", @@ -672,18 +673,18 @@ dependencies = [ [[package]] name = "crossbeam-channel" -version = "0.5.15" +version = "0.5.16" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "82b8f8f868b36967f9606790d1903570de9ceaf870a7bf9fbbd3016d636a2cb2" +checksum = "d85363c37faeca707aef026efa9f3b34d077bce547e48f770770625c6013679e" dependencies = [ "crossbeam-utils", ] [[package]] name = "crossbeam-deque" -version = "0.8.6" +version = "0.8.7" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9dd111b7b7f7d55b72c0a6ae361660ee5853c9af73f70c3c2ef6858b950e2e51" +checksum = "5181e0de7b61eb03a81e347d6dd8797bae9da5146707b51077e2d71a54ec0ceb" dependencies = [ "crossbeam-epoch", "crossbeam-utils", @@ -691,9 +692,9 @@ dependencies = [ [[package]] name = "crossbeam-epoch" -version = "0.9.18" +version = "0.9.20" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5b82ac4a3c2ca9c3460964f020e1402edd5753411d7737aa39c3714ad1b5420e" +checksum = "2d6914041f254d6e9176c01941b21115dcfb7089e55135a35411081bd106ef3f" dependencies = [ "crossbeam-utils", ] @@ -710,9 +711,9 @@ dependencies = [ [[package]] name = "crossbeam-utils" -version = "0.8.21" +version = "0.8.22" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d0a5c400df2834b80a4c3327b3aad3a4c4cd4de0629063962b03235697506a28" +checksum = "61803da095bee82a81bb1a452ecc25d3b2f1416d1897eb86430c6159ef717c17" [[package]] name = "crunchy" @@ -769,9 +770,6 @@ name = "deranged" version = "0.5.8" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "7cd812cc2bc1d69d4764bd80df88b4317eaef9e773c75226407d9bc0876b211c" -dependencies = [ - "powerfmt", -] [[package]] name = "digest" @@ -794,13 +792,13 @@ dependencies = [ [[package]] name = "displaydoc" -version = "0.2.5" +version = "0.2.7" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "97369cbbc041bc366949bc74d34658d6cda5621039731c6310521892a3a20ae0" +checksum = "c6232dd377dcc64799954cbd3a9bb882e9cdc1308ccd87b1c098f1fb2eaf82a8" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 3.0.4", ] [[package]] @@ -838,16 +836,16 @@ checksum = "f88959de2d447fd3eddcf1909d1f19fe084e27a056a6904203dc5d8b9e771c1e" dependencies = [ "rust_decimal", "serde", - "thiserror 2.0.18", + "thiserror 2.0.20", "time", "winnow 0.6.26", ] [[package]] name = "either" -version = "1.15.0" +version = "1.18.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "48c757948c5ede0e46177b7add2e67155f70e33c07fea8284df6576da70b3719" +checksum = "252afb9ae5eaa683babdc6a068b3f5726eb19e05070c731f9b2a23a7c3e8ed34" [[package]] name = "encode_unicode" @@ -872,7 +870,7 @@ checksum = "44f23cf4b44bfce11a86ace86f8a73ffdec849c9fd00a386a53d278bd9e81fb3" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] @@ -893,11 +891,10 @@ dependencies = [ [[package]] name = "event-listener" -version = "5.4.1" +version = "5.4.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e13b66accf52311f30a0db42147dadea9850cb48cd070028831ae5f5d4b856ab" +checksum = "5a23add41df1562121a9393cb065eab5146a1242410f23a644851e90cfd669d2" dependencies = [ - "concurrent-queue", "parking", "pin-project-lite", ] @@ -931,7 +928,7 @@ checksum = "c29b33a0187823f1fa88b36980227dc96c7504ede2288e7d2a77d9d6d88b260c" dependencies = [ "log", "once_cell", - "rand 0.9.4", + "rand 0.9.5", "tokio", ] @@ -947,9 +944,9 @@ dependencies = [ [[package]] name = "fastrand" -version = "2.4.1" +version = "2.5.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9f1f227452a390804cdb637b74a86990f2a7d7ba4b7d5693aac9b4dd6defd8d6" +checksum = "da7c62ceae207dd37ea5b845da6a0696c799f85e97da1ab5b7910be3c1c80223" [[package]] name = "figment" @@ -1009,7 +1006,7 @@ version = "25.12.19" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "35f6839d7b3b98adde531effaf34f0c2badc6f4735d26fe74709d8e513a96ef3" dependencies = [ - "bitflags 2.11.1", + "bitflags 2.13.1", "rustc_version", ] @@ -1029,12 +1026,6 @@ version = "1.0.7" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "3f9eec918d3f24069decb9af1554cad7c880e2da24a9afd88aca000531ab82c1" -[[package]] -name = "foldhash" -version = "0.1.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d9c4f5dac5e15c24eb999c26181a6ca40b39fe946cbe4c263c7209467bc83af2" - [[package]] name = "foldhash" version = "0.2.0" @@ -1104,7 +1095,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "db907f40a527ca2aa2f40a5f68b32ea58aa70f050cd233518e9ffd402cfba6ce" dependencies = [ "anyhow", - "bitflags 2.11.1", + "bitflags 2.13.1", "cmsketch", "equivalent", "foyer-common", @@ -1148,7 +1139,7 @@ dependencies = [ "mea", "parking_lot", "pin-project", - "rand 0.9.4", + "rand 0.9.5", "serde", "tracing", "twox-hash", @@ -1191,9 +1182,9 @@ checksum = "42703706b716c37f96a77aea830392ad231f44c9e9a67872fa5548707e11b11c" [[package]] name = "futures" -version = "0.3.32" +version = "0.3.33" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8b147ee9d1f6d097cef9ce628cd2ee62288d963e16fb287bd9286455b241382d" +checksum = "a88cf1f829d945f548cf8fec32c61b1f202b6d93b45848602fc02af4b12ad218" dependencies = [ "futures-channel", "futures-core", @@ -1206,9 +1197,9 @@ dependencies = [ [[package]] name = "futures-channel" -version = "0.3.32" +version = "0.3.33" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "07bbe89c50d7a535e539b8c17bc0b49bdb77747034daa8087407d655f3f7cc1d" +checksum = "262590f4fe6afeb0bc83be1daa64e52657fe185690a958af7f3ad0e92085c5ae" dependencies = [ "futures-core", "futures-sink", @@ -1216,15 +1207,15 @@ dependencies = [ [[package]] name = "futures-core" -version = "0.3.32" +version = "0.3.33" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7e3450815272ef58cec6d564423f6e755e25379b217b0bc688e295ba24df6b1d" +checksum = "2cd50c473c80f6d7c3670a752354b8e569b1a7cbfdc0419ec88e5edad85e0dc7" [[package]] name = "futures-executor" -version = "0.3.32" +version = "0.3.33" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "baf29c38818342a3b26b5b923639e7b1f4a61fc5e76102d4b1981c6dc7a7579d" +checksum = "6754879cc9f2c66f88c6e5c35344bb0bdb0708b0352b1201815667c7eabc7458" dependencies = [ "futures-core", "futures-task", @@ -1233,44 +1224,44 @@ dependencies = [ [[package]] name = "futures-io" -version = "0.3.32" +version = "0.3.33" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cecba35d7ad927e23624b22ad55235f2239cfa44fd10428eecbeba6d6a717718" +checksum = "4577ecaa3c4f96589d473f679a71b596316f6641bc350038b962a5daf0085d7a" [[package]] name = "futures-macro" -version = "0.3.32" +version = "0.3.33" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e835b70203e41293343137df5c0664546da5745f82ec9b84d40be8336958447b" +checksum = "2d6d3cde68c518367be28956066ddfef33813991b77a55005a69dae04bf3b10b" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] name = "futures-sink" -version = "0.3.32" +version = "0.3.33" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c39754e157331b013978ec91992bde1ac089843443c49cbc7f46150b0fad0893" +checksum = "e34418ac499d6305c2fb5ad0ed2f6ac998c5f8ca209b4510f7f94242c647e307" [[package]] name = "futures-task" -version = "0.3.32" +version = "0.3.33" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "037711b3d59c33004d3856fbdc83b99d4ff37a24768fa1be9ce3538a1cde4393" +checksum = "b231ed28831efb4a61a08580c4bc233ec56bc009f4cd8f52da2c3cb97df0c109" [[package]] name = "futures-timer" -version = "3.0.3" +version = "3.0.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f288b0a4f20f9a56b5d1da57e2227c661b7b16168e2f72365f57b63326e29b24" +checksum = "af43fadb8a98512d547e37b4e92e0ced13e205c061b87b4623eff01d918d6968" [[package]] name = "futures-util" -version = "0.3.32" +version = "0.3.33" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "389ca41296e6190b48053de0321d02a77f32f8a5d2461dd38762c0593805c6d6" +checksum = "a77a90a256fce34da66415271e30f94ee91c57b04b8a2c042d9cf3220179deaa" dependencies = [ "futures-channel", "futures-core", @@ -1313,25 +1304,23 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "899def5c37c4fd7b2664648c28120ecec138e4d395b459e5ca34f9cce2dd77fd" dependencies = [ "cfg-if", - "js-sys", "libc", "r-efi 5.3.0", "wasip2", - "wasm-bindgen", ] [[package]] name = "getrandom" -version = "0.4.2" +version = "0.4.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0de51e6874e94e7bf76d726fc5d13ba782deca734ff60d5bb2fb2607c7406555" +checksum = "300e883d756b2e4ec94e02791f39b04b522276138852cfc41d9fb7e904106099" dependencies = [ "cfg-if", + "js-sys", "libc", "r-efi 6.0.0", "rand_core 0.10.1", - "wasip2", - "wasip3", + "wasm-bindgen", ] [[package]] @@ -1342,9 +1331,9 @@ checksum = "e629b9b98ef3dd8afe6ca2bd0f89306cec16d43d907889945bc5d6687f2f13c7" [[package]] name = "glob" -version = "0.3.3" +version = "0.3.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0cc23270f6e1808e30a928bdc84dea0b9b4136a8bc82338574f23baf47bbd280" +checksum = "e4eba85ea1d0a966a983acd07deee566e67395d2d96b6fb39e62b5a833f1eb0b" [[package]] name = "gloo-timers" @@ -1371,9 +1360,9 @@ dependencies = [ [[package]] name = "h2" -version = "0.4.14" +version = "0.4.15" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "171fefbc92fe4a4de27e0698d6a5b392d6a0e333506bc49133760b3bcf948733" +checksum = "6cb093c84e8bd9b188d4c4a8cb6579fc016968d14c99882163cd3ff402a4f155" dependencies = [ "atomic-waker", "bytes", @@ -1404,9 +1393,6 @@ name = "hashbrown" version = "0.15.5" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "9229cfe53dfd69f0609a49f65461bd93001ea1ef889cd5529dd176593f5338a1" -dependencies = [ - "foldhash 0.1.5", -] [[package]] name = "hashbrown" @@ -1416,7 +1402,7 @@ checksum = "841d1cc9bed7f9236f321df977030373f4a4163ae1a7dbfe1a51a2c1a51d9100" dependencies = [ "allocator-api2", "equivalent", - "foldhash 0.2.0", + "foldhash", ] [[package]] @@ -1427,7 +1413,7 @@ checksum = "ed5909b6e89a2db4456e54cd5f673791d7eca6732202bbf2a9cc504fe2f9b84a" dependencies = [ "allocator-api2", "equivalent", - "foldhash 0.2.0", + "foldhash", ] [[package]] @@ -1450,9 +1436,9 @@ checksum = "fc0fef456e4baa96da950455cd02c081ca953b141298e41db3fc7e36b1da849c" [[package]] name = "http" -version = "1.4.0" +version = "1.5.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e3ba2a386d7f85a81f119ad7498ebe444d2e22c2af0b86b069416ace48b3311a" +checksum = "918d3568bebf352712bc2ef3d46a8bcf1a75b373be6539de198e9105cbbf9ce0" dependencies = [ "bytes", "itoa", @@ -1460,9 +1446,9 @@ dependencies = [ [[package]] name = "http-body" -version = "1.0.1" +version = "1.1.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1efedce1fb8e6913f23e0c92de8e62cd5b772a67e7b3946df930a62566c93184" +checksum = "ca2a8f2913ee65f60facd6a5905613afaa448497a0230cc41ce022d93290bc2c" dependencies = [ "bytes", "http", @@ -1470,9 +1456,9 @@ dependencies = [ [[package]] name = "http-body-util" -version = "0.1.3" +version = "0.1.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b021d93e26becf5dc7e1b75b1bed1fd93124b374ceb73f43d4d4eafec896a64a" +checksum = "e9f41fd6a08e4d4ec69df65976da761afd5ad5e58a9d4acb46bd1c953a9e3ff2" dependencies = [ "bytes", "futures-core", @@ -1489,24 +1475,24 @@ checksum = "6dbf3de79e51f3d586ab4cb9d5c3e2c14aa28ed23d180cf89b4df0454a69cc87" [[package]] name = "humantime" -version = "2.3.0" +version = "2.4.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "135b12329e5e3ce057a9f972339ea52bc954fe1e9358ef27f95e89716fbc5424" +checksum = "15cdd26707701c53297e2fa6afb323d55fbc1d0810c3aec078ae3ef0424c3c15" [[package]] name = "hybrid-array" -version = "0.4.12" +version = "0.4.14" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9155a582abd142abc056962c29e3ce5ff2ad5469f4246b537ed42c5deba857da" +checksum = "707114b52a152fa7bdb290cd7cd5912d9467273b6d74e21b8d81aca1f8533f6b" dependencies = [ "typenum", ] [[package]] name = "hyper" -version = "1.9.0" +version = "1.11.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6299f016b246a94207e63da54dbe807655bf9e00044f73ded42c3ac5305fbcca" +checksum = "d22053281f852e11534f5198498373cbb59295120a20771d90f7ed1897490a72" dependencies = [ "atomic-waker", "bytes", @@ -1573,7 +1559,7 @@ dependencies = [ "js-sys", "log", "wasm-bindgen", - "windows-core", + "windows-core 0.62.2", ] [[package]] @@ -1667,12 +1653,6 @@ dependencies = [ "zerovec", ] -[[package]] -name = "id-arena" -version = "2.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3d3067d79b975e8844ca9eb072e16b31c3c1c36928edf9c6789548c524d0d954" - [[package]] name = "idna" version = "1.1.0" @@ -1714,9 +1694,9 @@ checksum = "c8fae54786f62fb2918dcfae3d568594e50eb9b5c25bf04371af6fe7516452fb" [[package]] name = "insta" -version = "1.47.2" +version = "1.48.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7b4a6248eb93a4401ed2f37dfe8ea592d3cf05b7cf4f8efa867b6895af7e094e" +checksum = "86f0f8fee8c926415c58d6ae43a08523a26faccb2323f5e6b644fe7dd4ef6b82" dependencies = [ "console", "once_cell", @@ -1726,20 +1706,20 @@ dependencies = [ [[package]] name = "io-uring" -version = "0.7.12" +version = "0.7.13" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4d09b98f7eace8982db770e4408e7470b028ce513ac28fecdc6bf4c30fe92b62" +checksum = "9080b15e63775b9a2ac7dca720f7050a8b955e092ea0f6020a4a80f69998cdc0" dependencies = [ - "bitflags 2.11.1", + "bitflags 2.13.1", "cfg-if", "libc", ] [[package]] name = "ipnet" -version = "2.12.0" +version = "2.12.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d98f6fed1fde3f8c21bc40a1abb88dd75e67924f9cffc3ef95607bad8017f8e2" +checksum = "6a756c3fac73139e83f14c2d742155dd2b78d3ee56597b419a0579b7bdd6dd78" [[package]] name = "is-terminal" @@ -1803,7 +1783,7 @@ dependencies = [ "jni-sys", "log", "simd_cesu8", - "thiserror 2.0.18", + "thiserror 2.0.20", "walkdir", "windows-link 0.2.1", ] @@ -1818,7 +1798,7 @@ dependencies = [ "quote", "rustc_version", "simd_cesu8", - "syn", + "syn 2.0.119", ] [[package]] @@ -1837,28 +1817,27 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "38c0b942f458fe50cdac086d2f946512305e5631e720728f2a61aabcd47a6264" dependencies = [ "quote", - "syn", + "syn 2.0.119", ] [[package]] name = "jobserver" -version = "0.1.34" +version = "0.1.35" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9afb3de4395d6b3e67a780b6de64b51c978ecf11cb9a462c66be7d4ca9039d33" +checksum = "1c00acbd29eabad4a2392fa0e921c874934dbbf4194312ad20f04a0ed67a3cb3" dependencies = [ - "getrandom 0.3.4", + "getrandom 0.4.3", "libc", ] [[package]] name = "js-sys" -version = "0.3.98" +version = "0.3.103" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "67df7112613f8bfd9150013a0314e196f4800d3201ae742489d999db2f979f08" +checksum = "53b44bfcdb3f8d5837a46dae1ca9660a837176eee74a28b229bc626816589102" dependencies = [ "cfg-if", "futures-util", - "once_cell", "wasm-bindgen", ] @@ -1868,17 +1847,11 @@ version = "1.5.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "bbd2bcb4c963f2ddae06a2efc7e9f3591312473c50c6685e1f298068316e66fe" -[[package]] -name = "leb128fmt" -version = "0.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "09edd9e8b54e49e587e4f6295a7d29c3ea94d469cb40ab8ca70b288248a81db2" - [[package]] name = "libc" -version = "0.2.186" +version = "0.2.189" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "68ab91017fe16c622486840e4c83c9a37afeff978bd239b5293d61ece587de66" +checksum = "3eaf3ede3fee6db1a4c2ee091bf8a8b4dccdc6d17f656fb07896ee72867612f2" [[package]] name = "linux-raw-sys" @@ -1903,15 +1876,15 @@ dependencies = [ [[package]] name = "log" -version = "0.4.29" +version = "0.4.33" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5e5032e24019045c762d3c0f28f5b6b8bbf38563a65908389bf7978758920897" +checksum = "0ceec5bc11778974d1bcb055b18002eba7f4b3518b6a0081b3af5f21666da9ad" [[package]] name = "lru" -version = "0.18.0" +version = "0.18.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8a860605968fce16869fd239cf4237a82f3ac470723415db603b0e8b6c8d4fb9" +checksum = "5d2f2f9b4ba7e6b24d95e7e899329d35be83bcded72c8540cdd5368932d1d90a" dependencies = [ "hashbrown 0.17.1", ] @@ -1971,24 +1944,24 @@ dependencies = [ [[package]] name = "mea" -version = "0.6.3" +version = "0.6.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6747f54621d156e1b47eb6b25f39a941b9fc347f98f67d25d8881ff99e8ed832" +checksum = "31fc7d159de0085ab6dd7ff145a9819442cfd3d098f783263120503c3f3e58b0" dependencies = [ "slab", ] [[package]] name = "memchr" -version = "2.8.0" +version = "2.8.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f8ca58f447f06ed17d5fc4043ce1b10dd205e060fb3ce5b979b8ed8e59ff3f79" +checksum = "cf8baf1c55e62ffcace7a9f06f4bd9cd3f0c4beb022d3b367256b91b87513d98" [[package]] name = "memmap2" -version = "0.9.10" +version = "0.9.11" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "714098028fe011992e1c3962653c96b2d578c4b4bce9036e15ff220319b1e0e3" +checksum = "d1219ed1b7f229ee7104d281dd01d6802fe28bb6e95d292942c4daacdeb798c0" dependencies = [ "libc", ] @@ -2020,9 +1993,9 @@ dependencies = [ [[package]] name = "mio" -version = "1.2.0" +version = "1.2.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "50b7e5b27aa02a74bac8c3f23f448f8d87ff11f92d3aac1a6ed369ee08cc56c1" +checksum = "30d65c71f1ce40ab09135ce117d742b9f8a19ff91a41a8b57ed50bc2de59c427" dependencies = [ "libc", "wasi", @@ -2031,11 +2004,11 @@ dependencies = [ [[package]] name = "mixtrics" -version = "0.2.3" +version = "0.2.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fb252c728b9d77c6ef9103f0c81524fa0a3d3b161d0a936295d7fbeff6e04c11" +checksum = "2c46b5adfb7a3ae4996d327a5bdc90e78fec025806dd312bdbe6f07a755e0ec9" dependencies = [ - "itertools 0.14.0", + "itertools 0.15.0", "parking_lot", ] @@ -2076,7 +2049,7 @@ version = "0.31.3" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "cf20d2fde8ff38632c426f1165ed7436270b44f199fc55284c38276f9db47c3d" dependencies = [ - "bitflags 2.11.1", + "bitflags 2.13.1", "cfg-if", "cfg_aliases", "libc", @@ -2141,7 +2114,7 @@ version = "0.3.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "2a180dd8642fa45cdb7dd721cd4c11b1cadd4929ce112ebd8b9f5803cc79d536" dependencies = [ - "bitflags 2.11.1", + "bitflags 2.13.1", ] [[package]] @@ -2165,9 +2138,9 @@ dependencies = [ [[package]] name = "object_store" -version = "0.14.0" +version = "0.14.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "765784b4390c6bcf80316e5a22f4e3661b639c9d8c83246856643c27d8ce9dbe" +checksum = "d354792e39fa5f0009e47623cf8b15b099bf9a652fa55c6f817fe28ac84fea50" dependencies = [ "async-trait", "aws-lc-rs", @@ -2190,13 +2163,13 @@ dependencies = [ "parking_lot", "percent-encoding", "quick-xml", - "rand 0.10.1", + "rand 0.10.2", "reqwest", "rustls-pki-types", "serde", "serde_json", "serde_urlencoded", - "thiserror 2.0.18", + "thiserror 2.0.20", "tokio", "tracing", "url", @@ -2251,7 +2224,7 @@ dependencies = [ "proc-macro2", "proc-macro2-diagnostics", "quote", - "syn", + "syn 2.0.119", ] [[package]] @@ -2311,7 +2284,7 @@ dependencies = [ "proc-macro2", "proc-macro2-diagnostics", "quote", - "syn", + "syn 2.0.119", ] [[package]] @@ -2347,7 +2320,7 @@ checksum = "c96395f0a926bc13b1c17622aaddda1ecb55d49c8f1bf9777e4d877800a43f8b" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] @@ -2398,9 +2371,9 @@ dependencies = [ [[package]] name = "portable-atomic" -version = "1.13.1" +version = "1.15.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c33a9471896f1c69cecef8d20cbe2f7accd12527ce60845ff44c153bb2a21b49" +checksum = "05c8b63e8d9609db387f0324918f81d68fe27748f084ef092fb35954d0539a85" [[package]] name = "potential_utf" @@ -2450,16 +2423,6 @@ dependencies = [ "zerocopy", ] -[[package]] -name = "prettyplease" -version = "0.2.37" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "479ca8adacdd7ce8f1fb39ce9ecccbfe93a3f1344b3d0d97f20bc0196208f62b" -dependencies = [ - "proc-macro2", - "syn", -] - [[package]] name = "proc-macro-crate" version = "3.5.0" @@ -2471,9 +2434,9 @@ dependencies = [ [[package]] name = "proc-macro2" -version = "1.0.106" +version = "1.0.107" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8fd00f0bb2e90d81d1044c2b32617f68fcb9fa3bb7640c23e9c748e53fb30934" +checksum = "985e7ec9bb745e6ce6535b544d84d6cd6f7ad8bd711c398938ae983b91a766d9" dependencies = [ "unicode-ident", ] @@ -2486,7 +2449,7 @@ checksum = "af066a9c399a26e020ada66a034357a868728e72cd426f3adcd35f80d88d88c8" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.119", "version_check", "yansi", ] @@ -2499,9 +2462,9 @@ checksum = "4b45fcc2344c680f5025fe57779faef368840d0bd1f42f216291f0dc4ace4744" dependencies = [ "bit-set", "bit-vec", - "bitflags 2.11.1", + "bitflags 2.13.1", "num-traits", - "rand 0.9.4", + "rand 0.9.5", "rand_chacha", "rand_xorshift", "regex-syntax", @@ -2543,9 +2506,9 @@ checksum = "a1d01941d82fa2ab50be1e79e6714289dd7cde78eba4c074bc5a4374f650dfe0" [[package]] name = "quick-xml" -version = "0.40.1" +version = "0.41.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2474bd2e5029e7ccb6abb2ba48cf2383a333851dedf495901544281590c7da7f" +checksum = "e660451e55124f798a69a5af3f49ccfbefbd41910eefd25caf2393e1f3473ec1" dependencies = [ "memchr", "serde", @@ -2553,9 +2516,9 @@ dependencies = [ [[package]] name = "quinn" -version = "0.11.9" +version = "0.11.11" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b9e20a958963c291dc322d98411f541009df2ced7b5a4f2bd52337638cfccf20" +checksum = "0c1a41e437b6bbd489372cd4971de128e85c855f56c57f283d20ff016cf7c0a8" dependencies = [ "bytes", "cfg_aliases", @@ -2565,7 +2528,7 @@ dependencies = [ "rustc-hash", "rustls", "socket2", - "thiserror 2.0.18", + "thiserror 2.0.20", "tokio", "tracing", "web-time", @@ -2573,21 +2536,22 @@ dependencies = [ [[package]] name = "quinn-proto" -version = "0.11.14" +version = "0.11.16" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "434b42fec591c96ef50e21e886936e66d3cc3f737104fdb9b737c40ffb94c098" +checksum = "2f4bfc015262b9df63c8845072ce59068853ff5872180c2ce2f13038b970e560" dependencies = [ "aws-lc-rs", "bytes", - "getrandom 0.3.4", + "getrandom 0.4.3", "lru-slab", - "rand 0.9.4", + "rand 0.10.2", + "rand_pcg", "ring", "rustc-hash", "rustls", "rustls-pki-types", "slab", - "thiserror 2.0.18", + "thiserror 2.0.20", "tinyvec", "tracing", "web-time", @@ -2595,23 +2559,23 @@ dependencies = [ [[package]] name = "quinn-udp" -version = "0.5.14" +version = "0.5.15" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "addec6a0dcad8a8d96a771f815f0eaf55f9d1805756410b39f5fa81332574cbd" +checksum = "35a133f956daabe89a61a685c2649f13d82d5aa4bd5d12d1277e1072a21c0694" dependencies = [ "cfg_aliases", "libc", "once_cell", "socket2", "tracing", - "windows-sys 0.59.0", + "windows-sys 0.61.2", ] [[package]] name = "quote" -version = "1.0.45" +version = "1.0.47" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "41f2619966050689382d2b44f664f4bc593e129785a36d6ee376ddf37259b924" +checksum = "1fbf4db142a473a8d80c26bbf18454ed458bf8d26c8219c331daecfdbd079001" dependencies = [ "proc-macro2", ] @@ -2630,9 +2594,9 @@ checksum = "f8dcc9c7d52a811697d2151c701e0d08956f92b0e24136cf4cf27b57a6a0d9bf" [[package]] name = "rand" -version = "0.9.4" +version = "0.9.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "44c5af06bb1b7d3216d91932aed5265164bf384dc89cd6ba05cf59a35f5f76ea" +checksum = "b9ef1d0d795eb7d84685bca4f72f3649f064e6641543d3a8c415898726a57b41" dependencies = [ "rand_chacha", "rand_core 0.9.5", @@ -2640,12 +2604,12 @@ dependencies = [ [[package]] name = "rand" -version = "0.10.1" +version = "0.10.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d2e8e8bcc7961af1fdac401278c6a831614941f6164ee3bf4ce61b7edb162207" +checksum = "c7f5fa3a058cd35567ef9bfa5e75732bee0f9e4c55fa90477bef2dfcdbc4be80" dependencies = [ "chacha20", - "getrandom 0.4.2", + "getrandom 0.4.3", "rand_core 0.10.1", ] @@ -2674,6 +2638,15 @@ version = "0.10.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "63b8176103e19a2643978565ca18b50549f6101881c443590420e4dc998a3c69" +[[package]] +name = "rand_pcg" +version = "0.10.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "caa0f4137e1c0a72f4c651489402276c8e8e1cf081f3b0ba156d2cbeef09e86a" +dependencies = [ + "rand_core 0.10.1", +] + [[package]] name = "rand_xorshift" version = "0.4.0" @@ -2718,14 +2691,14 @@ version = "0.5.18" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "ed2bf2547551a7053d6fdfafda3f938979645c44812fbfcda098faae3f1a362d" dependencies = [ - "bitflags 2.11.1", + "bitflags 2.13.1", ] [[package]] name = "regex" -version = "1.12.3" +version = "1.13.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e10754a14b9137dd7b1e3e5b0493cc9171fdd105e0ab477f51b72e7f3ac0e276" +checksum = "f020237b6c8eed93db2e2cb53c00c60a8e1bc73da7d073199a1180401450218d" dependencies = [ "aho-corasick", "memchr", @@ -2735,9 +2708,9 @@ dependencies = [ [[package]] name = "regex-automata" -version = "0.4.14" +version = "0.4.18" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6e1dd4122fc1595e8162618945476892eefca7b88c52820e74af6262213cae8f" +checksum = "ad8553b9b26413251cbf30e620595c7a41b3887f03da04579c0e6b0d6a06b4b2" dependencies = [ "aho-corasick", "memchr", @@ -2746,9 +2719,9 @@ dependencies = [ [[package]] name = "regex-syntax" -version = "0.8.10" +version = "0.8.11" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "dc897dd8d9e8bd1ed8cdad82b5966c3e0ecae09fb1907d58efaa013543185d0a" +checksum = "d6f6ff9a378485b298a5286656da665ba74413d36db0979633275d2e708145d4" [[package]] name = "relative-path" @@ -2834,15 +2807,15 @@ dependencies = [ "regex", "relative-path", "rustc_version", - "syn", + "syn 2.0.119", "unicode-ident", ] [[package]] name = "rust_decimal" -version = "1.42.0" +version = "1.42.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0c5108e3d4d903e21aac27f12ba5377b6b34f9f44b325e4894c7924169d06995" +checksum = "be2a24f50780bc85f09cc6ac299bdf1424302742d77221106859c9d8b102126a" dependencies = [ "arrayvec", "num-traits", @@ -2850,15 +2823,15 @@ dependencies = [ [[package]] name = "rustc-demangle" -version = "0.1.27" +version = "0.1.28" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b50b8869d9fc858ce7266cce0194bd74df58b9d0e3f6df3a9fc8eb470d95c09d" +checksum = "b74b56ffa8bb2830709a538c2cbcae9aa062db0d2a42563bfb09bdaae44020eb" [[package]] name = "rustc-hash" -version = "2.1.2" +version = "2.1.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "94300abf3f1ae2e2b8ffb7b58043de3d399c73fa6f4b73826402a5c457614dbe" +checksum = "6b1e7f9a428571be2dc5bc0505c13fb6bf936822b894ec87abf8a08a4e51742d" [[package]] name = "rustc_version" @@ -2875,7 +2848,7 @@ version = "1.1.4" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "b6fe4565b9518b83ef4f91bb47ce29620ca828bd32cb7e408f0062e9930ba190" dependencies = [ - "bitflags 2.11.1", + "bitflags 2.13.1", "errno", "libc", "linux-raw-sys", @@ -2884,9 +2857,9 @@ dependencies = [ [[package]] name = "rustls" -version = "0.23.40" +version = "0.23.43" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ef86cd5876211988985292b91c96a8f2d298df24e75989a43a3c73f2d4d8168b" +checksum = "0283386ce02abc0151e1761d08802dfe86c173b0b494af5cbc086574e453da06" dependencies = [ "aws-lc-rs", "once_cell", @@ -2898,9 +2871,9 @@ dependencies = [ [[package]] name = "rustls-native-certs" -version = "0.8.3" +version = "0.8.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "612460d5f7bea540c490b2b6395d8e34a953e52b491accd6c86c8164c5932a63" +checksum = "dab5152771c58876a2146916e53e35057e1a4dfa2b9df0f0305b07f611fdea4d" dependencies = [ "openssl-probe", "rustls-pki-types", @@ -2910,9 +2883,9 @@ dependencies = [ [[package]] name = "rustls-pki-types" -version = "1.14.1" +version = "1.15.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "30a7197ae7eb376e574fe940d068c30fe0462554a3ddbe4eca7838e049c937a9" +checksum = "2f4925028c7eb5d1fcdaf196971378ed9d2c1c4efc7dc5d011256f76c99c0a96" dependencies = [ "web-time", "zeroize", @@ -2959,9 +2932,9 @@ dependencies = [ [[package]] name = "rustversion" -version = "1.0.22" +version = "1.0.23" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b39cdef0fa800fc44525c84ccb54a029961a8215f9619753635a9c0d2538d46d" +checksum = "cf54715a573b99ac80df0bc206da022bcd442c974952c7b9720069370852e21f" [[package]] name = "rusty-fork" @@ -3022,7 +2995,7 @@ checksum = "1783eabc414609e28a5ba76aee5ddd52199f7107a0b24c2e9746a1ecc34a683d" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] @@ -3031,7 +3004,7 @@ version = "3.7.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "b7f4bc775c73d9a02cde8bf7b2ec4c9d12743edf609006c7facc23998404cd1d" dependencies = [ - "bitflags 2.11.1", + "bitflags 2.13.1", "core-foundation", "core-foundation-sys", "libc", @@ -3060,9 +3033,9 @@ dependencies = [ [[package]] name = "serde" -version = "1.0.228" +version = "1.0.229" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9a8e94ea7f378bd32cbbd37198a4a91436180c5bb472411e48b5ec2e2124ae9e" +checksum = "4148590afebada386688f18773da617792bf2ef03ffc1e4cbd2b1d45b023e0ba" dependencies = [ "serde_core", "serde_derive", @@ -3070,29 +3043,29 @@ dependencies = [ [[package]] name = "serde_core" -version = "1.0.228" +version = "1.0.229" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "41d385c7d4ca58e59fc732af25c3983b67ac852c1a25000afe1175de458b67ad" +checksum = "67dca2c9c51e58a4791a4b1ed58308b39c64224d349a935ab5039aa360942a48" dependencies = [ "serde_derive", ] [[package]] name = "serde_derive" -version = "1.0.228" +version = "1.0.229" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d540f220d3187173da220f885ab66608367b6574e925011a9353e4badda91d79" +checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 3.0.4", ] [[package]] name = "serde_json" -version = "1.0.149" +version = "1.0.151" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "83fc039473c5595ace860d8c4fafa220ff474b3fc6bfdb4293327f1a37e94d86" +checksum = "c841b55ecdae098c80dcae9cf767f6f8a0c2cdb3416bbef72181df4d0fe73f14" dependencies = [ "itoa", "memchr", @@ -3155,9 +3128,9 @@ dependencies = [ [[package]] name = "shlex" -version = "1.3.0" +version = "2.0.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0fda2ff0d084019ba4d7c6f371c95d8fd75ce3524c3cb8fb653a3023f6323e64" +checksum = "f8fadd59c855ef2080decdef8ff161eb6661b86933c9d82e5ba29dc602a55aba" [[package]] name = "signal-hook-registry" @@ -3171,15 +3144,15 @@ dependencies = [ [[package]] name = "simd-adler32" -version = "0.3.9" +version = "0.3.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "703d5c7ef118737c72f1af64ad2f6f8c5e1921f818cdcb97b8fe6fc69bf66214" +checksum = "3a219298ac11a56ea9a6d2120044824d6f01aeb034955e7af7bc16858527deea" [[package]] name = "simd_cesu8" -version = "1.1.1" +version = "1.2.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "94f90157bb87cddf702797c5dadfa0be7d266cdf49e22da2fcaa32eff75b2c33" +checksum = "11031e251abf8611c80f460e19dbdeb54a66db918e49c65a7065b46ac7aec520" dependencies = [ "rustc_version", "simdutf8", @@ -3219,7 +3192,7 @@ dependencies = [ "backon", "backtrace", "bincode", - "bitflags 2.11.1", + "bitflags 2.13.1", "bytes", "chrono", "crc32fast", @@ -3246,7 +3219,7 @@ dependencies = [ "parking_lot", "pprof", "proptest", - "rand 0.9.4", + "rand 0.9.5", "rstest", "serde", "serde_json", @@ -3279,7 +3252,7 @@ dependencies = [ "clap", "futures", "object_store", - "rand 0.9.4", + "rand 0.9.5", "rand_xorshift", "slatedb", "sysinfo", @@ -3315,7 +3288,7 @@ dependencies = [ "chrono", "log", "object_store", - "rand 0.9.4", + "rand 0.9.5", "rand_xoshiro", "serde", "thread_local", @@ -3335,7 +3308,7 @@ dependencies = [ "log", "object_store", "parking_lot", - "rand 0.9.4", + "rand 0.9.5", "rstest", "slatedb", "slatedb-common", @@ -3393,27 +3366,27 @@ checksum = "88414a5ca1f85d82cc34471e975f0f74f6aa54c40f062efa42c0080e7f763f81" [[package]] name = "smallvec" -version = "1.15.1" +version = "1.15.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "67b1b7a3b5fe4f1376887184045fcf45c69e92af734b7aaddc05fb777b6fbd03" +checksum = "8ed6a63f02c8539c91a8685a86f4099661ba3da017932f6ebbea6de3f0fa7c90" [[package]] name = "smawk" -version = "0.3.2" +version = "0.3.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b7c388c1b5e93756d0c740965c41e8822f866621d41acbdf6336a6a168f8840c" +checksum = "e8e2fb0f499abb4d162f2bedad68f5ef91a1682b5a03596ddb67efd37768d100" [[package]] name = "snap" -version = "1.1.1" +version = "1.1.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1b6b67fb9a61334225b5b790716f609cd58395f895b3fe8b328786812a40bc3b" +checksum = "199905e6153d6405f9728fe44daace35f8f837bbf830bb6e85fbd5828709a886" [[package]] name = "socket2" -version = "0.6.3" +version = "0.6.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3a766e1110788c36f4fa1c2b71b387a7815aa65f88ce0229841826633d93723e" +checksum = "c3d1e2c7f27f8d4cb10542a02c49005dbd6e93095799d6f3be745fae9f8fedd4" dependencies = [ "libc", "windows-sys 0.61.2", @@ -3421,9 +3394,9 @@ dependencies = [ [[package]] name = "spin" -version = "0.10.0" +version = "0.10.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d5fe4ccb98d9c292d56fec89a5e07da7fc4cf0dc11e156b41793132775d3e591" +checksum = "023a211cb3138dbc438680b32560ad89f699977624c9f8dbb95a47d5b4c07dd3" dependencies = [ "lock_api", ] @@ -3477,9 +3450,20 @@ dependencies = [ [[package]] name = "syn" -version = "2.0.117" +version = "2.0.119" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e665b8803e7b1d2a727f4023456bbbbe74da67099c585258af0ad9c5013b9b99" +checksum = "872831b642d1a07999a962a351ed35b955ea2cfc8f3862091e2a240a84f17297" +dependencies = [ + "proc-macro2", + "quote", + "unicode-ident", +] + +[[package]] +name = "syn" +version = "3.0.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e6275cddf4610d1775e6d1fe9469b2e77d0f39fd98fb7450901b821e0c53649f" dependencies = [ "proc-macro2", "quote", @@ -3503,7 +3487,7 @@ checksum = "728a70f3dbaf5bab7f0c4b1ac8d7ae5ea60a4b5549c8a5914361c99147a709d2" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] @@ -3533,7 +3517,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "32497e9a4c7b38532efcdebeef879707aa9f794296a4f0244f6f69e9bc8574bd" dependencies = [ "fastrand", - "getrandom 0.4.2", + "getrandom 0.4.3", "once_cell", "rustix", "windows-sys 0.61.2", @@ -3559,11 +3543,11 @@ dependencies = [ [[package]] name = "thiserror" -version = "2.0.18" +version = "2.0.20" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4288b5bcbc7920c07a1149a35cf9590a2aa808e0bc1eafaade0b80947865fbc4" +checksum = "ec86235f5fcc2a73650310756d2ac5b138a5780bbbdfae3eeccec992c435ba4f" dependencies = [ - "thiserror-impl 2.0.18", + "thiserror-impl 2.0.20", ] [[package]] @@ -3574,34 +3558,34 @@ checksum = "4fee6c4efc90059e10f81e6d42c60a18f76588c3d74cb83a0b242a2b6c7504c1" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] name = "thiserror-impl" -version = "2.0.18" +version = "2.0.20" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ebc4ee7f67670e9b64d05fa4253e753e016c6c95ff35b89b7941d6b856dec1d5" +checksum = "bc04cd3e1236dd4a98afca4569f2deb3f120e5422a4023be2cb683f8486292af" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 3.0.4", ] [[package]] name = "thread_local" -version = "1.1.9" +version = "1.1.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f60246a4944f24f6e018aa17cdeffb7818b76356965d03b07d6a9886e8962185" +checksum = "1ad99c4c6d32803332c548b1af0540b357b3f5fc0be8f6c6bfe8b2e6ae784070" dependencies = [ "cfg-if", ] [[package]] name = "time" -version = "0.3.47" +version = "0.3.55" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "743bd48c283afc0388f9b8827b976905fb217ad9e647fae3a379a9283c4def2c" +checksum = "cdb87b95ec50ddfa440816d227a17b2ccbdda963a316a727fda0fc4334f7d134" dependencies = [ "deranged", "num-conv", @@ -3611,9 +3595,9 @@ dependencies = [ [[package]] name = "time-core" -version = "0.1.8" +version = "0.1.9" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7694e1cfe791f8d31026952abf09c69ca6f6fa4e1a1229e18988f06a04a12dca" +checksum = "9e1c906769ad99c88eaa54e728060edef082f8e358ff32030cb7c7d315e81109" [[package]] name = "tinystr" @@ -3637,9 +3621,9 @@ dependencies = [ [[package]] name = "tinyvec" -version = "1.11.0" +version = "1.12.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3e61e67053d25a4e82c844e8424039d9745781b3fc4f32b8d55ed50f5f667ef3" +checksum = "bb4ebadaa0af04fab11ae01eb5f9fdb5f9c5b875506e210e71c07873528baa7f" dependencies = [ "tinyvec_macros", ] @@ -3652,9 +3636,9 @@ checksum = "1f3ccbac311fea05f86f61904b462b55fb3df8837a366dfc601a0161d0532f20" [[package]] name = "tokio" -version = "1.52.3" +version = "1.53.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8fc7f01b389ac15039e4dc9531aa973a135d7a4135281b12d7c1bc79fd57fffe" +checksum = "202caea871b69668250d242070849eb495be178ed697a3e98aebce5bc81a0bed" dependencies = [ "bytes", "libc", @@ -3668,13 +3652,13 @@ dependencies = [ [[package]] name = "tokio-macros" -version = "2.7.0" +version = "2.7.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "385a6cb71ab9ab790c5fe8d67f1645e6c450a7ce006a33de03daa956cf70a496" +checksum = "78773a2a397f451582ce068015985c33193cf6dea8b74d2a639fe457b2f07b0e" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 3.0.4", ] [[package]] @@ -3689,9 +3673,9 @@ dependencies = [ [[package]] name = "tokio-stream" -version = "0.1.18" +version = "0.1.19" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "32da49809aab5c3bc678af03902d4ccddea2a87d028d86392a4b1560c6906c70" +checksum = "a3d06f0b082ba57c26b79407372e57cf2a1e28124f78e9479fe80322cf53420b" dependencies = [ "futures-core", "pin-project-lite", @@ -3711,15 +3695,16 @@ dependencies = [ [[package]] name = "tokio-util" -version = "0.7.18" +version = "0.7.19" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9ae9cec805b01e8fc3fd2fe289f89149a9b66dd16786abd8b19cfa7b48cb0098" +checksum = "494815d09bf52b5548659851081238f0ca39ff638363907596da739561c62c52" dependencies = [ "bytes", "futures-core", "futures-sink", "futures-util", "hashbrown 0.15.5", + "libc", "pin-project-lite", "tokio", ] @@ -3801,16 +3786,16 @@ dependencies = [ "indexmap", "toml_datetime 1.1.1+spec-1.1.0", "toml_parser", - "winnow 1.0.3", + "winnow 1.0.4", ] [[package]] name = "toml_parser" -version = "1.1.2+spec-1.1.0" +version = "1.1.3+spec-1.1.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a2abe9b86193656635d2411dc43050282ca48aa31c2451210f4202550afb7526" +checksum = "1d38ac1cf9b95face32296c0a3ede1fdc270627c9d9c02a7274dd6d960dc4d56" dependencies = [ - "winnow 1.0.3", + "winnow 1.0.4", ] [[package]] @@ -3821,9 +3806,9 @@ checksum = "5d99f8c9a7727884afe522e9bd5edbfc91a3312b36a77b5fb8926e4c31a41801" [[package]] name = "toml_writer" -version = "1.1.1+spec-1.1.0" +version = "1.1.2+spec-1.1.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "756daf9b1013ebe47a8776667b466417e2d4c5679d441c26230efd9ef78692db" +checksum = "7d56353a2a665ad0f41a421187180aab746c8c325620617ad883a99a1cbe66d2" [[package]] name = "tower" @@ -3846,7 +3831,7 @@ version = "0.6.11" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "4cfcf7e2740e6fc6d4d688b4ef00650406bb94adf4731e43c096c3a19fe40840" dependencies = [ - "bitflags 2.11.1", + "bitflags 2.13.1", "bytes", "futures-util", "http", @@ -3890,7 +3875,7 @@ checksum = "7490cfa5ec963746568740651ac6781f701c9c5ea257c58e057f3ba8cf69e8da" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] @@ -3940,18 +3925,18 @@ checksum = "e421abadd41a4225275504ea4d6566923418b7f05506fbc9c0fe86ba7396114b" [[package]] name = "twox-hash" -version = "2.1.2" +version = "2.1.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9ea3136b675547379c4bd395ca6b938e5ad3c3d20fad76e7fe85f9e0d011419c" +checksum = "8464ec13c3691491391d9fce00f6416c9a48e46972f72d7865688be2080192c9" dependencies = [ - "rand 0.9.4", + "rand 0.10.2", ] [[package]] name = "typenum" -version = "1.20.0" +version = "1.20.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "40ce102ab67701b8526c123c1bab5cbe42d7040ccfd0f64af1a385808d2f43de" +checksum = "b6f5e870be6c3b371b77fe0ee0bafb859fa4964b4404c27de1d380043c4dda20" [[package]] name = "ulid" @@ -3959,7 +3944,7 @@ version = "1.2.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "470dbf6591da1b39d43c14523b2b469c86879a53e8b758c8e090a470fe7b1fbe" dependencies = [ - "rand 0.9.4", + "rand 0.9.5", "serde", "web-time", ] @@ -3985,17 +3970,11 @@ version = "1.0.24" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "e6e4313cd5fcd3dad5cafa179702e2b244f760991f45397d14d4ebf38247da75" -[[package]] -name = "unicode-xid" -version = "0.2.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ebc1c04c71510c7f702b52b7c350734c9ff1295c464a03335b00bb84fc54f853" - [[package]] name = "uniffi" -version = "0.31.1" +version = "0.31.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "dc5f2297ee5b893405bed1a6929faec4713a061df158ecf5198089f23910d470" +checksum = "46eefd5468602930da46b1f49d3448c6dfc2e81295f93120f23f8174fd70267f" dependencies = [ "anyhow", "cargo_metadata", @@ -4007,9 +3986,9 @@ dependencies = [ [[package]] name = "uniffi_bindgen" -version = "0.31.1" +version = "0.31.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8bc0c60a9607e7ab77a2ad47ec5530178015014839db25af7512447d2238016c" +checksum = "c4a0c9b375d32e1365cdb2bdd7cb495eecf6fac851ddbad077412b4ee1888514" dependencies = [ "anyhow", "askama", @@ -4033,9 +4012,9 @@ dependencies = [ [[package]] name = "uniffi_core" -version = "0.31.1" +version = "0.31.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "77baf5d539fe2e1ad6805e942dbc5dbdeb2b83eb5f2b3a6535d422ca4b02a12f" +checksum = "eec017b112701681f6fbbe5d92014b5c468eb0b177a94389de03ceec40665095" dependencies = [ "anyhow", "async-compat", @@ -4046,22 +4025,22 @@ dependencies = [ [[package]] name = "uniffi_internal_macros" -version = "0.31.1" +version = "0.31.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b4b42137524f4be6400fcaca9d02c1d4ecb6ad917e4013c0b93235526d8396e5" +checksum = "4641669b48fefbc5e80ff08c5004d9c7617fb91232131a6734ab6712779cb04c" dependencies = [ "anyhow", "indexmap", "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] name = "uniffi_macros" -version = "0.31.1" +version = "0.31.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d9273ec45330d8fe9a3701b7b983cea7a4e218503359831967cb95d26b873561" +checksum = "eeb8617ee814de22caf7417bf514715ba0b3f46bd9d5a5d794413fd8282cb737" dependencies = [ "camino", "fs-err", @@ -4069,16 +4048,16 @@ dependencies = [ "proc-macro2", "quote", "serde", - "syn", + "syn 2.0.119", "toml 0.9.12+spec-1.1.0", "uniffi_meta", ] [[package]] name = "uniffi_meta" -version = "0.31.1" +version = "0.31.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "431d2f443e7828a6c29d188de98b6771a6491ee98bba2d4372643bf93f988a18" +checksum = "58d5b94fc92803d21b2928bd15c6f06e57609b95caf98ea561c99cda1b6d2a25" dependencies = [ "anyhow", "siphasher", @@ -4088,9 +4067,9 @@ dependencies = [ [[package]] name = "uniffi_pipeline" -version = "0.31.1" +version = "0.31.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "761ef74f6175e15603d0424cc5f98854c5baccfe7bf4ccb08e5816f9ab8af689" +checksum = "032739b3ec725576914c15899dedaf080163ced86b6934566c20ec2b20ce90ca" dependencies = [ "anyhow", "heck 0.5.0", @@ -4101,9 +4080,9 @@ dependencies = [ [[package]] name = "uniffi_udl" -version = "0.31.1" +version = "0.31.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "68773ec0e1c067b6505a73bbf6a5782f31a7f9209333a0df97b87565c46bf370" +checksum = "fc0a1d0a0252ce1af9e8ce78ba67ac0d8937fb2bedaf10cbddd43d3614d06ec6" dependencies = [ "anyhow", "textwrap", @@ -4149,11 +4128,11 @@ checksum = "06abde3611657adf66d383f00b093d7faecc7fa57071cce2578660c9f1010821" [[package]] name = "uuid" -version = "1.23.1" +version = "1.24.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ddd74a9687298c6858e9b88ec8935ec45d22e8fd5e6394fa1bd4e99a87789c76" +checksum = "bf3923a6f5c4c6382e0b653c4117f48d631ea17f38ed86e2a828e6f7412f5239" dependencies = [ - "getrandom 0.4.2", + "getrandom 0.4.3", "js-sys", "serde_core", "wasm-bindgen", @@ -4207,27 +4186,18 @@ checksum = "ccf3ec651a847eb01de73ccad15eb7d99f80485de043efb2f370cd654f4ea44b" [[package]] name = "wasip2" -version = "1.0.3+wasi-0.2.9" +version = "1.0.4+wasi-0.2.12" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "20064672db26d7cdc89c7798c48a0fdfac8213434a1186e5ef29fd560ae223d6" +checksum = "b67efb37e106e55ce722a510d6b5f9c17f083e5fc79afc2badeb12cc313d9487" dependencies = [ - "wit-bindgen 0.57.1", -] - -[[package]] -name = "wasip3" -version = "0.4.0+wasi-0.3.0-rc-2026-01-06" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5428f8bf88ea5ddc08faddef2ac4a67e390b88186c703ce6dbd955e1c145aca5" -dependencies = [ - "wit-bindgen 0.51.0", + "wit-bindgen", ] [[package]] name = "wasm-bindgen" -version = "0.2.121" +version = "0.2.126" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "49ace1d07c165b0864824eee619580c4689389afa9dc9ed3a4c75040d82e6790" +checksum = "4b067c0c11094aef6b7a801c1e34a26affafdf3d051dba08456b868789aaf9a4" dependencies = [ "cfg-if", "once_cell", @@ -4238,9 +4208,9 @@ dependencies = [ [[package]] name = "wasm-bindgen-futures" -version = "0.4.71" +version = "0.4.76" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "96492d0d3ffba25305a7dc88720d250b1401d7edca02cc3bcd50633b424673b8" +checksum = "c62df1340f32221cb9c54d6a27b030e3dba64361d4a95bed55f9aacb44da291d" dependencies = [ "js-sys", "wasm-bindgen", @@ -4248,9 +4218,9 @@ dependencies = [ [[package]] name = "wasm-bindgen-macro" -version = "0.2.121" +version = "0.2.126" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8e68e6f4afd367a562002c05637acb8578ff2dea1943df76afb9e83d177c8578" +checksum = "167ce5e579f6bcf889c4f7175a8a5a585de84e8ff93976ce393efa5f2837aab1" dependencies = [ "quote", "wasm-bindgen-macro-support", @@ -4258,48 +4228,26 @@ dependencies = [ [[package]] name = "wasm-bindgen-macro-support" -version = "0.2.121" +version = "0.2.126" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d95a9ec35c64b2a7cb35d3fead40c4238d0940c86d107136999567a4703259f2" +checksum = "f3997c7839262f4ef12cf90b818d6340c18e80f263f1a94bf157d0ec4420380e" dependencies = [ "bumpalo", "proc-macro2", "quote", - "syn", + "syn 2.0.119", "wasm-bindgen-shared", ] [[package]] name = "wasm-bindgen-shared" -version = "0.2.121" +version = "0.2.126" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c4e0100b01e9f0d03189a92b96772a1fb998639d981193d7dbab487302513441" +checksum = "dc1b4cb0cc549fcf58d7dfc081778139b3d283a081644e833e84682ad71cea24" dependencies = [ "unicode-ident", ] -[[package]] -name = "wasm-encoder" -version = "0.244.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "990065f2fe63003fe337b932cfb5e3b80e0b4d0f5ff650e6985b1048f62c8319" -dependencies = [ - "leb128fmt", - "wasmparser", -] - -[[package]] -name = "wasm-metadata" -version = "0.244.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bb0e353e6a2fbdc176932bbaab493762eb1255a7900fe0fea1a2f96c296cc909" -dependencies = [ - "anyhow", - "indexmap", - "wasm-encoder", - "wasmparser", -] - [[package]] name = "wasm-streams" version = "0.5.0" @@ -4313,23 +4261,11 @@ dependencies = [ "web-sys", ] -[[package]] -name = "wasmparser" -version = "0.244.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "47b807c72e1bac69382b3a6fb3dbe8ea4c0ed87ff5629b8685ae6b9a611028fe" -dependencies = [ - "bitflags 2.11.1", - "hashbrown 0.15.5", - "indexmap", - "semver", -] - [[package]] name = "web-sys" -version = "0.3.98" +version = "0.3.103" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4b572dff8bcf38bad0fa19729c89bb5748b2b9b1d8be70cf90df697e3a8f32aa" +checksum = "8622dcb61c0bcc9fffa6938bed81210af2da9a7e4a1a834b2e37a59b6dfb6141" dependencies = [ "js-sys", "wasm-bindgen", @@ -4347,9 +4283,9 @@ dependencies = [ [[package]] name = "webpki-root-certs" -version = "1.0.8" +version = "1.0.9" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0d46a5a140e6f7afeccd8eae97eff335163939eac8b929834875168b29b3d267" +checksum = "b96554aa2acc8ccdb7e1c9a58a7a68dd5d13bccc69cd124cb09406db612a1c9b" dependencies = [ "rustls-pki-types", ] @@ -4401,7 +4337,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "9babd3a767a4c1aef6900409f85f5d53ce2544ccdfaa86dad48c91782c6d6893" dependencies = [ "windows-collections", - "windows-core", + "windows-core 0.61.2", "windows-future", "windows-link 0.1.3", "windows-numerics", @@ -4413,7 +4349,7 @@ version = "0.2.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "3beeceb5e5cfd9eb1d76b381630e82c4241ccd0d27f1a39ed41b2760b255c5e8" dependencies = [ - "windows-core", + "windows-core 0.61.2", ] [[package]] @@ -4425,8 +4361,21 @@ dependencies = [ "windows-implement", "windows-interface", "windows-link 0.1.3", - "windows-result", - "windows-strings", + "windows-result 0.3.4", + "windows-strings 0.4.2", +] + +[[package]] +name = "windows-core" +version = "0.62.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b8e83a14d34d0623b51dce9581199302a221863196a1dde71a7663a4c2be9deb" +dependencies = [ + "windows-implement", + "windows-interface", + "windows-link 0.2.1", + "windows-result 0.4.1", + "windows-strings 0.5.1", ] [[package]] @@ -4435,7 +4384,7 @@ version = "0.2.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "fc6a41e98427b19fe4b73c550f060b59fa592d7d686537eebf9385621bfbad8e" dependencies = [ - "windows-core", + "windows-core 0.61.2", "windows-link 0.1.3", "windows-threading", ] @@ -4448,7 +4397,7 @@ checksum = "053e2e040ab57b9dc951b72c264860db7eb3b0200ba345b4e4c3b14f67855ddf" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] @@ -4459,7 +4408,7 @@ checksum = "3f316c4a2570ba26bbec722032c4099d8c8bc095efccdc15688708623367e358" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] @@ -4480,7 +4429,7 @@ version = "0.2.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "9150af68066c4c5c07ddc0ce30421554771e528bde427614c61038bc2c92c2b1" dependencies = [ - "windows-core", + "windows-core 0.61.2", "windows-link 0.1.3", ] @@ -4493,6 +4442,15 @@ dependencies = [ "windows-link 0.1.3", ] +[[package]] +name = "windows-result" +version = "0.4.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7781fa89eaf60850ac3d2da7af8e5242a5ea78d1a11c49bf2910bb5a73853eb5" +dependencies = [ + "windows-link 0.2.1", +] + [[package]] name = "windows-strings" version = "0.4.2" @@ -4502,6 +4460,15 @@ dependencies = [ "windows-link 0.1.3", ] +[[package]] +name = "windows-strings" +version = "0.5.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7837d08f69c77cf6b07689544538e017c1bfcf57e34b4c0ff58e6c2cd3b37091" +dependencies = [ + "windows-link 0.2.1", +] + [[package]] name = "windows-sys" version = "0.52.0" @@ -4622,107 +4589,19 @@ dependencies = [ [[package]] name = "winnow" -version = "1.0.3" +version = "1.0.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0592e1c9d151f854e6fd382574c3a0855250e1d9b2f99d9281c6e6391af352f1" +checksum = "23b97319f7b8343df12cc98938e5c3eb436064524c8d2b4e30a1d3a36eecdf81" dependencies = [ "memchr", ] -[[package]] -name = "wit-bindgen" -version = "0.51.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d7249219f66ced02969388cf2bb044a09756a083d0fab1e566056b04d9fbcaa5" -dependencies = [ - "wit-bindgen-rust-macro", -] - [[package]] name = "wit-bindgen" version = "0.57.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "1ebf944e87a7c253233ad6766e082e3cd714b5d03812acc24c318f549614536e" -[[package]] -name = "wit-bindgen-core" -version = "0.51.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ea61de684c3ea68cb082b7a88508a8b27fcc8b797d738bfc99a82facf1d752dc" -dependencies = [ - "anyhow", - "heck 0.5.0", - "wit-parser", -] - -[[package]] -name = "wit-bindgen-rust" -version = "0.51.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b7c566e0f4b284dd6561c786d9cb0142da491f46a9fbed79ea69cdad5db17f21" -dependencies = [ - "anyhow", - "heck 0.5.0", - "indexmap", - "prettyplease", - "syn", - "wasm-metadata", - "wit-bindgen-core", - "wit-component", -] - -[[package]] -name = "wit-bindgen-rust-macro" -version = "0.51.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0c0f9bfd77e6a48eccf51359e3ae77140a7f50b1e2ebfe62422d8afdaffab17a" -dependencies = [ - "anyhow", - "prettyplease", - "proc-macro2", - "quote", - "syn", - "wit-bindgen-core", - "wit-bindgen-rust", -] - -[[package]] -name = "wit-component" -version = "0.244.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9d66ea20e9553b30172b5e831994e35fbde2d165325bec84fc43dbf6f4eb9cb2" -dependencies = [ - "anyhow", - "bitflags 2.11.1", - "indexmap", - "log", - "serde", - "serde_derive", - "serde_json", - "wasm-encoder", - "wasm-metadata", - "wasmparser", - "wit-parser", -] - -[[package]] -name = "wit-parser" -version = "0.244.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ecc8ac4bc1dc3381b7f59c34f00b67e18f910c2c0f50015669dde7def656a736" -dependencies = [ - "anyhow", - "id-arena", - "indexmap", - "log", - "semver", - "serde", - "serde_derive", - "serde_json", - "unicode-xid", - "wasmparser", -] - [[package]] name = "writeable" version = "0.6.3" @@ -4737,9 +4616,9 @@ checksum = "cfe53a6657fd280eaa890a3bc59152892ffa3e30101319d168b781ed6529b049" [[package]] name = "yoke" -version = "0.8.2" +version = "0.8.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "abe8c5fda708d9ca3df187cae8bfb9ceda00dd96231bed36e445a1a48e66f9ca" +checksum = "709fe23a0424b6a435d82152b1bd3fdfb0833487d5fa90d05d42762a9891fef5" dependencies = [ "stable_deref_trait", "yoke-derive", @@ -4754,28 +4633,28 @@ checksum = "de844c262c8848816172cef550288e7dc6c7b7814b4ee56b3e1553f275f1858e" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.119", "synstructure", ] [[package]] name = "zerocopy" -version = "0.8.48" +version = "0.8.56" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "eed437bf9d6692032087e337407a86f04cd8d6a16a37199ed57949d415bd68e9" +checksum = "556764e583adb45a9f8d413c2a147fa7e8d821e48e12b14fd560b607998b75eb" dependencies = [ "zerocopy-derive", ] [[package]] name = "zerocopy-derive" -version = "0.8.48" +version = "0.8.56" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "70e3cd084b1788766f53af483dd21f93881ff30d7320490ec3ef7526d203bad4" +checksum = "f2ab42fc20575779bd240faa45f94a74256f755c0fa9e89f0ede20d91d0cdfc1" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] @@ -4795,15 +4674,15 @@ checksum = "11532158c46691caf0f2593ea8358fed6bbf68a0315e80aae9bd41fbade684a1" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.119", "synstructure", ] [[package]] name = "zeroize" -version = "1.8.2" +version = "1.9.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b97154e67e32c85465826e8bcc1c59429aaaf107c1e4a9e53c8d8ccd5eff88d0" +checksum = "e13c156562582aa81c60cb29407084cdb54c4164760106ab78e6c5b0858cf64e" [[package]] name = "zerotrie" @@ -4835,14 +4714,14 @@ checksum = "625dc425cab0dca6dc3c3319506e6593dcb08a9f387ea3b284dbd52a92c40555" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] name = "zmij" -version = "1.0.21" +version = "1.0.23" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b8848ee67ecc8aedbaf3e4122217aff892639231befc6a1b58d29fff4c2cabaa" +checksum = "29666d0abbfad1e3dc4dcf6144730dd3a3ab225bbbdac83319345b1b44ccfc1b" [[package]] name = "zstd" diff --git a/slatedb/benches/db_reader_memory_scaling.rs b/slatedb/benches/db_reader_memory_scaling.rs index baa8ecaa59..ae75b9f09d 100644 --- a/slatedb/benches/db_reader_memory_scaling.rs +++ b/slatedb/benches/db_reader_memory_scaling.rs @@ -232,10 +232,7 @@ async fn write_wal(db: &Db, index: usize, segmentation: Segmentation) { &key_for(index, segmentation), b"value-value-value-value-value", &PutOptions::default(), - &WriteOptions { - await_durable: false, - ..WriteOptions::default() - }, + &WriteOptions::default(), ) .await .expect("put failed"); diff --git a/slatedb/benches/db_reader_scaling.rs b/slatedb/benches/db_reader_scaling.rs index c55c23c4e8..d686e08403 100644 --- a/slatedb/benches/db_reader_scaling.rs +++ b/slatedb/benches/db_reader_scaling.rs @@ -175,10 +175,7 @@ async fn write_wal(db: &Db, index: usize, segmentation: Segmentation) -> Bytes { &key, b"value-value-value-value-value", &PutOptions::default(), - &WriteOptions { - await_durable: false, - ..WriteOptions::default() - }, + &WriteOptions::default(), ) .await .expect("put failed"); @@ -498,10 +495,7 @@ async fn benchmark_bulk_iterator(full: bool) { &key, b"value", &PutOptions::default(), - &WriteOptions { - await_durable: false, - ..WriteOptions::default() - }, + &WriteOptions::default(), ) .await .expect("put failed"); @@ -593,10 +587,7 @@ async fn benchmark_iterator_contention(full: bool) { &key_for(index, Segmentation::None), b"value", &PutOptions::default(), - &WriteOptions { - await_durable: false, - ..WriteOptions::default() - }, + &WriteOptions::default(), ) .await .expect("put failed"); @@ -1043,10 +1034,7 @@ async fn build_pinned_generations( &key, b"value", &PutOptions::default(), - &WriteOptions { - await_durable: false, - ..WriteOptions::default() - }, + &WriteOptions::default(), ) .await .expect("put failed"); diff --git a/slatedb/src/cached_object_store/storage_fs.rs b/slatedb/src/cached_object_store/storage_fs.rs index 87de02df96..6ebf5d9974 100644 --- a/slatedb/src/cached_object_store/storage_fs.rs +++ b/slatedb/src/cached_object_store/storage_fs.rs @@ -1493,7 +1493,7 @@ mod tests { .prefix("objstore_cache_usage_snapshot_") .tempdir() .unwrap(); - let recorder = slatedb_common::metrics::MetricsRecorderHelper::noop(); + let recorder = MetricsRecorderHelper::noop(); let usage = Arc::new(FsCacheUsage { initialized: AtomicBool::new(false), used_bytes: AtomicU64::new(0), @@ -1567,7 +1567,7 @@ mod tests { .prefix("objstore_cache_usage_updates_") .tempdir() .unwrap(); - let recorder = slatedb_common::metrics::MetricsRecorderHelper::noop(); + let recorder = MetricsRecorderHelper::noop(); let storage = FsCacheStorage::new( temp_dir.path().to_path_buf(), None, diff --git a/slatedb/src/db_reader.rs b/slatedb/src/db_reader.rs index 0dccc5feed..6023a8a809 100644 --- a/slatedb/src/db_reader.rs +++ b/slatedb/src/db_reader.rs @@ -957,10 +957,7 @@ impl DbReaderInner { /// The `(last replayed WAL id, last committed seq)` the reader has already /// reached: the watermark of the most recently replayed table, or the /// manifest's own boundary when nothing has been replayed into `tables`. - fn replayed_watermark( - core: &ManifestCore, - tables: &ReplayMemtables, - ) -> (u64, u64) { + fn replayed_watermark(core: &ManifestCore, tables: &ReplayMemtables) -> (u64, u64) { match tables.front() { Some(latest_replayed_table) => ( latest_replayed_table.recent_flushed_wal_id(), @@ -982,8 +979,8 @@ impl DbReaderInner { mut publish: Option>, db_stats: Option<&DbStats>, ) -> Result<(u64, u64), SlateDBError> { - let (mut replay_after_wal_id, mut last_committed_seq) = replay_cursor - .unwrap_or_else(|| Self::replayed_watermark(core, into_tables)); + let (mut replay_after_wal_id, mut last_committed_seq) = + replay_cursor.unwrap_or_else(|| Self::replayed_watermark(core, into_tables)); let wal_id_start = replay_after_wal_id .checked_add(1) .ok_or(SlateDBError::InvalidDBState)?; @@ -2086,7 +2083,7 @@ mod tests { }, db_cache::{test_utils::TestCache, DbCache}, db_reader::{DbReader, DbReaderInner, DbReaderMode, DbReaderOptions}, - db_state::{SsTableId, SstType}, + db_state::SstType, db_stats::DbStats, db_status::DbStatusManager, dispatcher::MessageHandler, @@ -3079,10 +3076,7 @@ mod tests { .await .unwrap(); - let write_options = WriteOptions { - await_durable: false, - ..WriteOptions::default() - }; + let write_options = WriteOptions::default(); db.put_with_options(b"key", b"v1", &PutOptions::default(), &write_options) .await .unwrap(); @@ -3166,10 +3160,7 @@ mod tests { .build() .await .unwrap(); - let write_options = WriteOptions { - await_durable: false, - ..WriteOptions::default() - }; + let write_options = WriteOptions::default(); db.put_with_options(b"key", b"v1", &PutOptions::default(), &write_options) .await .unwrap(); @@ -3278,10 +3269,7 @@ mod tests { .build() .await .unwrap(); - let write_options = WriteOptions { - await_durable: false, - ..WriteOptions::default() - }; + let write_options = WriteOptions::default(); db.put_with_options(b"key", b"v1", &PutOptions::default(), &write_options) .await .unwrap(); @@ -3438,10 +3426,7 @@ mod tests { }) .await .unwrap(); - let write_options = WriteOptions { - await_durable: false, - ..WriteOptions::default() - }; + let write_options = WriteOptions::default(); db.put_with_options(b"key", b"v1", &PutOptions::default(), &write_options) .await .unwrap(); @@ -3542,10 +3527,7 @@ mod tests { }) .await .unwrap(); - let write_options = WriteOptions { - await_durable: false, - ..WriteOptions::default() - }; + let write_options = WriteOptions::default(); db.put_with_options(b"a", b"old", &PutOptions::default(), &write_options) .await .unwrap(); @@ -4054,9 +4036,11 @@ mod tests { let reader = DbReader::open_internal( test_provider.manifest_store(), test_provider.table_store(), + test_provider.wal_store(), DbReaderMode::ManagedCheckpoint, None, None, + None, reader_options, Arc::clone(&test_provider.system_clock), Arc::clone(&test_provider.rand), @@ -5105,10 +5089,7 @@ mod tests { b"key", b"value", &PutOptions::default(), - &WriteOptions { - await_durable: false, - ..WriteOptions::default() - }, + &WriteOptions::default(), ) .await .unwrap(); diff --git a/slatedb/src/db_transaction.rs b/slatedb/src/db_transaction.rs index ced0f32bf5..354524805d 100644 --- a/slatedb/src/db_transaction.rs +++ b/slatedb/src/db_transaction.rs @@ -228,7 +228,7 @@ impl DbTransaction { let db_state = self.db_inner.state.read().view(); - let mut key_to_idx = std::collections::HashMap::::with_capacity(keys.len()); + let mut key_to_idx = HashMap::::with_capacity(keys.len()); let mut unique_keys = Vec::::with_capacity(keys.len()); let mut output_positions = Vec::>::with_capacity(keys.len()); diff --git a/slatedb/src/error.rs b/slatedb/src/error.rs index d7fd2f259f..9a6a8dd82e 100644 --- a/slatedb/src/error.rs +++ b/slatedb/src/error.rs @@ -828,7 +828,7 @@ mod tests { #[test] fn database_missing_has_stable_code_without_changing_broad_kind() { - let public_err = Error::from(SlateDBError::DatabaseMissing); + let public_err = Error::from(DatabaseMissing); assert_eq!(public_err.kind(), ErrorKind::Data); assert_eq!(public_err.code(), Some(ErrorCode::DatabaseMissing)); @@ -837,7 +837,7 @@ mod tests { #[test] fn other_data_errors_do_not_claim_database_is_missing() { for err in [ - SlateDBError::LatestTransactionalObjectVersionMissing, + LatestTransactionalObjectVersionMissing, SlateDBError::ManifestMissing(7), SlateDBError::InvalidDBState, ] { From 2b8f979e431694919abe3f2763abb637b9173dfb Mon Sep 17 00:00:00 2001 From: Matthew Sanetra <41018997+matthewsanetra@users.noreply.github.com> Date: Tue, 1 Sep 2026 17:28:46 +0100 Subject: [PATCH 64/65] Finish v0.16 CI and consumer compatibility --- bindings/go/uniffi/slatedb.go | 5 + bindings/go/uniffi/slatedb_test.go | 12 +- .../java/io/slatedb/uniffi/SlateDbDbTest.java | 6 +- bindings/node/tests/support.mjs | 3 +- bindings/uniffi/src/config.rs | 15 +- bindings/uniffi/src/settings.rs | 1 + examples/src/azure_blob_storage.rs | 1 + examples/src/google_cloud_storage.rs | 1 + slatedb-dst/src/actors/bank/mod.rs | 5 +- slatedb-dst/src/actors/bank/transfer.rs | 1 + slatedb-dst/src/actors/workload.rs | 4 +- .../src/deterministic_local_filesystem.rs | 2 +- slatedb/benches/db_operations.rs | 1 + slatedb/benches/db_reader_memory_scaling.rs | 5 +- slatedb/benches/db_reader_scaling.rs | 20 +- slatedb/benches/db_transaction.rs | 1 + slatedb/benches/scan_prefix_bench.rs | 5 +- slatedb/src/admin.rs | 6 +- slatedb/src/batch_write.rs | 2 + .../src/cached_object_store/object_store.rs | 4 +- slatedb/src/checkpoint.rs | 14 +- slatedb/src/clone.rs | 2 + slatedb/src/compaction_execute_bench.rs | 2 +- slatedb/src/compactor.rs | 60 +++- slatedb/src/compactor_state.rs | 4 +- slatedb/src/config.rs | 19 +- slatedb/src/db.rs | 256 ++++++++++++++---- slatedb/src/db_cache_manager.rs | 1 + slatedb/src/db_reader.rs | 55 +++- slatedb/src/db_snapshot.rs | 1 + slatedb/src/db_state.rs | 2 +- slatedb/src/db_transaction.rs | 27 +- slatedb/src/format/block.rs | 4 +- slatedb/src/format/block_v2.rs | 4 +- slatedb/src/mem_table.rs | 4 +- .../src/memtable_flusher/manifest_writer.rs | 5 +- slatedb/src/ops.rs | 28 +- slatedb/src/paths.rs | 4 +- slatedb/src/segment_iterator.rs | 2 +- slatedb/src/test_utils.rs | 18 +- slatedb/src/transaction_manager.rs | 2 +- slatedb/src/wal/slatedb/iterator.rs | 2 +- slatedb/tests/db.rs | 3 + slatedb/tests/prefix_filter.rs | 20 +- 44 files changed, 490 insertions(+), 149 deletions(-) diff --git a/bindings/go/uniffi/slatedb.go b/bindings/go/uniffi/slatedb.go index 0bfad5b4e1..00e0e49a38 100644 --- a/bindings/go/uniffi/slatedb.go +++ b/bindings/go/uniffi/slatedb.go @@ -11287,11 +11287,14 @@ func (_ FfiDestroyerWalRows) Destroy(value WalRows) { // Options that control writes and commits. type WriteOptions struct { + // Whether the call waits for the write to become durable before returning. + AwaitDurable bool // Optional caller-supplied sequence number. Zero uses SlateDB's sequence oracle. Seqnum uint64 } func (r *WriteOptions) Destroy() { + FfiDestroyerBool{}.Destroy(r.AwaitDurable) FfiDestroyerUint64{}.Destroy(r.Seqnum) } @@ -11305,6 +11308,7 @@ func (c FfiConverterWriteOptions) Lift(rb RustBufferI) WriteOptions { func (c FfiConverterWriteOptions) Read(reader io.Reader) WriteOptions { return WriteOptions{ + FfiConverterBoolINSTANCE.Read(reader), FfiConverterUint64INSTANCE.Read(reader), } } @@ -11318,6 +11322,7 @@ func (c FfiConverterWriteOptions) LowerExternal(value WriteOptions) ExternalCRus } func (c FfiConverterWriteOptions) Write(writer io.Writer, value WriteOptions) { + FfiConverterBoolINSTANCE.Write(writer, value.AwaitDurable) FfiConverterUint64INSTANCE.Write(writer, value.Seqnum) } diff --git a/bindings/go/uniffi/slatedb_test.go b/bindings/go/uniffi/slatedb_test.go index b05ad1d267..4e677dab43 100644 --- a/bindings/go/uniffi/slatedb_test.go +++ b/bindings/go/uniffi/slatedb_test.go @@ -679,7 +679,7 @@ func TestDbCrudAndMetadata(t *testing.T) { } putOptions := slatedb.PutOptions{Ttl: slatedb.TtlDefault{}} - writeOptions := slatedb.WriteOptions{Seqnum: 0} + writeOptions := slatedb.WriteOptions{AwaitDurable: true, Seqnum: 0} firstWrite, err := handle.db.Put([]byte("alpha"), []byte("one")) if err != nil { @@ -965,7 +965,7 @@ func TestDbBatchWriteAndConsumption(t *testing.T) { t.Fatalf("WriteBatch.PutWithOptions(): %v", err) } - secondBatchWrite, err := handle.db.WriteWithOptions(secondBatch, slatedb.WriteOptions{Seqnum: 0}) + secondBatchWrite, err := handle.db.WriteWithOptions(secondBatch, slatedb.WriteOptions{AwaitDurable: true, Seqnum: 0}) if err != nil { t.Fatalf("WriteWithOptions(): %v", err) } @@ -1032,7 +1032,7 @@ func TestDbMerge(t *testing.T) { []byte("merge"), []byte(":two"), slatedb.MergeOptions{Ttl: slatedb.TtlDefault{}}, - slatedb.WriteOptions{Seqnum: 0}, + slatedb.WriteOptions{AwaitDurable: true, Seqnum: 0}, ) if err != nil { t.Fatalf("MergeWithOptions(): %v", err) @@ -3135,7 +3135,7 @@ func TestDbTtl(t *testing.T) { key, value := []byte("alpha"), []byte("one") putOptions := slatedb.PutOptions{Ttl: slatedb.TtlExpireAtMillis{Field0: 1}} - writeOptions := slatedb.WriteOptions{Seqnum: 0} + writeOptions := slatedb.WriteOptions{AwaitDurable: true, Seqnum: 0} write, err := handle.db.PutWithOptions(key, value, putOptions, writeOptions) if err != nil { t.Fatalf("Put(alpha): %v", err) @@ -3195,7 +3195,7 @@ const batchSeedTtlMillis = 3_600_000 func seedBatchRows(t *testing.T, db *slatedb.Db) { t.Helper() - writeOptions := slatedb.WriteOptions{Seqnum: 0} + writeOptions := slatedb.WriteOptions{AwaitDurable: true, Seqnum: 0} for _, row := range batchSeedRows { putOptions := slatedb.PutOptions{Ttl: row.ttl} write, err := db.PutWithOptions([]byte(row.key), []byte(row.value), putOptions, writeOptions) @@ -3454,7 +3454,7 @@ func openBenchDB(b *testing.B) *slatedb.Db { db.Destroy() }) - writeOptions := slatedb.WriteOptions{Seqnum: 0} + writeOptions := slatedb.WriteOptions{AwaitDurable: false, Seqnum: 0} putOptions := slatedb.PutOptions{Ttl: slatedb.TtlDefault{}} for i := 0; i < benchScanRows; i++ { key := []byte(fmt.Sprintf("bench:%06d", i)) diff --git a/bindings/java/slatedb-uniffi/src/test/java/io/slatedb/uniffi/SlateDbDbTest.java b/bindings/java/slatedb-uniffi/src/test/java/io/slatedb/uniffi/SlateDbDbTest.java index e226e67670..ae8f3ab5cb 100644 --- a/bindings/java/slatedb-uniffi/src/test/java/io/slatedb/uniffi/SlateDbDbTest.java +++ b/bindings/java/slatedb-uniffi/src/test/java/io/slatedb/uniffi/SlateDbDbTest.java @@ -94,7 +94,7 @@ void dbCrudAndMetadata() throws Exception { ReadOptions readOptions = TestSupport.readOptions(); PutOptions putOptions = new PutOptions(new Ttl.Default()); - WriteOptions writeOptions = new WriteOptions(0L); + WriteOptions writeOptions = new WriteOptions(true, 0L); WriteHandle firstWrite = TestSupport.await(db.put(TestSupport.bytes("alpha"), TestSupport.bytes("one"))); assertNotNull(firstWrite); @@ -268,7 +268,7 @@ void dbBatchWriteAndConsumption() throws Exception { new PutOptions(new Ttl.Default())); try (WriteHandle writeHandle = - TestSupport.await(db.writeWithOptions(secondBatch, new WriteOptions(0L)))) { + TestSupport.await(db.writeWithOptions(secondBatch, new WriteOptions(true, 0L)))) { TestSupport.await(writeHandle.awaitDurable()); } } @@ -309,7 +309,7 @@ void dbMerge() throws Exception { TestSupport.bytes("merge"), TestSupport.bytes(":two"), new MergeOptions(new Ttl.Default()), - new WriteOptions(0L)))) { + new WriteOptions(true, 0L)))) { TestSupport.await(writeHandle.awaitDurable()); } assertArrayEquals( diff --git a/bindings/node/tests/support.mjs b/bindings/node/tests/support.mjs index 1cbd3f56a0..4d91b0197b 100644 --- a/bindings/node/tests/support.mjs +++ b/bindings/node/tests/support.mjs @@ -69,8 +69,9 @@ export function readerOptions(skipWalReplay) { }; } -export function writeOptions() { +export function writeOptions(awaitDurable = true) { return { + await_durable: awaitDurable, seqnum: 0, }; } diff --git a/bindings/uniffi/src/config.rs b/bindings/uniffi/src/config.rs index 1ed64e0898..0b28fafe81 100644 --- a/bindings/uniffi/src/config.rs +++ b/bindings/uniffi/src/config.rs @@ -330,16 +330,29 @@ impl TryFrom for slatedb::config::ScanOptions { } /// Options that control writes and commits. -#[derive(Clone, Debug, Default, uniffi::Record)] +#[derive(Clone, Debug, uniffi::Record)] pub struct WriteOptions { + /// Whether the call waits for the write to become durable before returning. + #[uniffi(default = true)] + pub await_durable: bool, /// Optional caller-supplied sequence number. Zero uses SlateDB's sequence oracle. #[uniffi(default = 0)] pub seqnum: u64, } +impl Default for WriteOptions { + fn default() -> Self { + Self { + await_durable: true, + seqnum: 0, + } + } +} + impl From for slatedb::config::WriteOptions { fn from(value: WriteOptions) -> Self { slatedb::config::WriteOptions { + await_durable: value.await_durable, seqnum: value.seqnum, } } diff --git a/bindings/uniffi/src/settings.rs b/bindings/uniffi/src/settings.rs index d17975f86b..76e13974eb 100644 --- a/bindings/uniffi/src/settings.rs +++ b/bindings/uniffi/src/settings.rs @@ -169,6 +169,7 @@ fn apply_dotted_json_path(root: &mut Value, key: &str, value: Value) -> Result<( } #[cfg(test)] +#[allow(clippy::result_large_err)] mod tests { use serde_json::json; use std::sync::Arc; diff --git a/examples/src/azure_blob_storage.rs b/examples/src/azure_blob_storage.rs index 135db85d53..4efc2604d6 100644 --- a/examples/src/azure_blob_storage.rs +++ b/examples/src/azure_blob_storage.rs @@ -25,6 +25,7 @@ async fn main() -> anyhow::Result<()> { // Put 1000 keys, do not wait for it to be durable println!("Writing 1000 keys without waiting for flush"); let write_options = slatedb::config::WriteOptions { + await_durable: false, ..Default::default() }; for i in 0..1000 { diff --git a/examples/src/google_cloud_storage.rs b/examples/src/google_cloud_storage.rs index f24fe87bd9..11ede8629e 100644 --- a/examples/src/google_cloud_storage.rs +++ b/examples/src/google_cloud_storage.rs @@ -24,6 +24,7 @@ async fn main() -> anyhow::Result<()> { // Put 1000 keys, do not wait for it to be durable println!("Writing 1000 keys without waiting for flush"); let write_options = slatedb::config::WriteOptions { + await_durable: false, ..Default::default() }; for i in 0..1000 { diff --git a/slatedb-dst/src/actors/bank/mod.rs b/slatedb-dst/src/actors/bank/mod.rs index df303b1dec..6b6b76154f 100644 --- a/slatedb-dst/src/actors/bank/mod.rs +++ b/slatedb-dst/src/actors/bank/mod.rs @@ -1,6 +1,8 @@ mod auditor; mod transfer; +use std::mem::size_of; + use bytes::Bytes; use slatedb::config::{PutOptions, WriteOptions}; use slatedb::{Db, DbTransaction, Error, MergeOperator, MergeOperatorError}; @@ -8,7 +10,7 @@ use slatedb::{Db, DbTransaction, Error, MergeOperator, MergeOperatorError}; pub use self::auditor::{AuditorActor, BankAuditView}; pub use self::transfer::{TransferActor, TransferMode}; -const ACCUMULATOR_BYTES: usize = std::mem::size_of::(); +const ACCUMULATOR_BYTES: usize = size_of::(); /// Configuration for the deterministic bank workload. #[derive(Clone, Debug)] @@ -86,6 +88,7 @@ pub async fn initialize_accounts(db: &Db, options: &BankOptions) -> Result<(), E &starting_balance, &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) diff --git a/slatedb-dst/src/actors/bank/transfer.rs b/slatedb-dst/src/actors/bank/transfer.rs index 5c82b2d402..39515be947 100644 --- a/slatedb-dst/src/actors/bank/transfer.rs +++ b/slatedb-dst/src/actors/bank/transfer.rs @@ -40,6 +40,7 @@ impl TransferActor { impl Actor for TransferActor { async fn run(&mut self, ctx: &ActorCtx) -> Result<(), Error> { let write_options = WriteOptions { + await_durable: false, ..WriteOptions::default() }; let from_rand = ctx.rand().rng().next_u64(); diff --git a/slatedb-dst/src/actors/workload.rs b/slatedb-dst/src/actors/workload.rs index 52e1615aae..cb78dda322 100644 --- a/slatedb-dst/src/actors/workload.rs +++ b/slatedb-dst/src/actors/workload.rs @@ -1,4 +1,5 @@ use std::collections::{BTreeMap, BTreeSet}; +use std::mem::size_of; use std::sync::atomic::{AtomicU64, Ordering}; use async_trait::async_trait; @@ -13,7 +14,7 @@ use crate::{utils::build_scan_options, Actor, ActorCtx}; use super::PROGRESS_LOG_INTERVAL; -const WORKLOAD_VALUE_VERSION_SIZE: usize = std::mem::size_of::(); +const WORKLOAD_VALUE_VERSION_SIZE: usize = size_of::(); /// Configuration for the mixed DST workload actor. #[derive(Clone, Debug)] @@ -160,6 +161,7 @@ impl Actor for WorkloadActor { async fn run(&mut self, ctx: &ActorCtx) -> Result<(), Error> { let put_options = PutOptions::default(); let write_options = WriteOptions { + await_durable: false, ..WriteOptions::default() }; let key_prefix = self diff --git a/slatedb-dst/src/deterministic_local_filesystem.rs b/slatedb-dst/src/deterministic_local_filesystem.rs index 13ffc7fb85..4b53e9fe06 100644 --- a/slatedb-dst/src/deterministic_local_filesystem.rs +++ b/slatedb-dst/src/deterministic_local_filesystem.rs @@ -1030,7 +1030,7 @@ async fn read_range_with_yields_from_file( Ok(buffer.into()) } fn convert_walkdir_result( - result: std::result::Result, + result: Result, ) -> object_store::Result> { match result { Ok(entry) => match symlink_metadata(entry.path()) { diff --git a/slatedb/benches/db_operations.rs b/slatedb/benches/db_operations.rs index e37c4ec4e2..6f1055ccba 100644 --- a/slatedb/benches/db_operations.rs +++ b/slatedb/benches/db_operations.rs @@ -25,6 +25,7 @@ fn criterion_benchmark(c: &mut Criterion) { value, &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) diff --git a/slatedb/benches/db_reader_memory_scaling.rs b/slatedb/benches/db_reader_memory_scaling.rs index ae75b9f09d..baa8ecaa59 100644 --- a/slatedb/benches/db_reader_memory_scaling.rs +++ b/slatedb/benches/db_reader_memory_scaling.rs @@ -232,7 +232,10 @@ async fn write_wal(db: &Db, index: usize, segmentation: Segmentation) { &key_for(index, segmentation), b"value-value-value-value-value", &PutOptions::default(), - &WriteOptions::default(), + &WriteOptions { + await_durable: false, + ..WriteOptions::default() + }, ) .await .expect("put failed"); diff --git a/slatedb/benches/db_reader_scaling.rs b/slatedb/benches/db_reader_scaling.rs index d686e08403..c55c23c4e8 100644 --- a/slatedb/benches/db_reader_scaling.rs +++ b/slatedb/benches/db_reader_scaling.rs @@ -175,7 +175,10 @@ async fn write_wal(db: &Db, index: usize, segmentation: Segmentation) -> Bytes { &key, b"value-value-value-value-value", &PutOptions::default(), - &WriteOptions::default(), + &WriteOptions { + await_durable: false, + ..WriteOptions::default() + }, ) .await .expect("put failed"); @@ -495,7 +498,10 @@ async fn benchmark_bulk_iterator(full: bool) { &key, b"value", &PutOptions::default(), - &WriteOptions::default(), + &WriteOptions { + await_durable: false, + ..WriteOptions::default() + }, ) .await .expect("put failed"); @@ -587,7 +593,10 @@ async fn benchmark_iterator_contention(full: bool) { &key_for(index, Segmentation::None), b"value", &PutOptions::default(), - &WriteOptions::default(), + &WriteOptions { + await_durable: false, + ..WriteOptions::default() + }, ) .await .expect("put failed"); @@ -1034,7 +1043,10 @@ async fn build_pinned_generations( &key, b"value", &PutOptions::default(), - &WriteOptions::default(), + &WriteOptions { + await_durable: false, + ..WriteOptions::default() + }, ) .await .expect("put failed"); diff --git a/slatedb/benches/db_transaction.rs b/slatedb/benches/db_transaction.rs index 1bc7aab609..5d26400f13 100644 --- a/slatedb/benches/db_transaction.rs +++ b/slatedb/benches/db_transaction.rs @@ -65,6 +65,7 @@ fn merge_options() -> MergeOptions { fn write_options() -> WriteOptions { WriteOptions { + await_durable: false, ..WriteOptions::default() } } diff --git a/slatedb/benches/scan_prefix_bench.rs b/slatedb/benches/scan_prefix_bench.rs index a85bbd7b86..93ca65241f 100644 --- a/slatedb/benches/scan_prefix_bench.rs +++ b/slatedb/benches/scan_prefix_bench.rs @@ -136,7 +136,10 @@ fn recency_scan_options() -> ScanOptions { } async fn populate(db: &Db) { - let write_opts = WriteOptions::default(); + let write_opts = WriteOptions { + await_durable: false, + ..WriteOptions::default() + }; let put_opts = PutOptions::default(); let mut next_version = [0u64; NUM_PREFIXES]; for round in 0..NUM_FLUSHES { diff --git a/slatedb/src/admin.rs b/slatedb/src/admin.rs index 040705c9e5..aef082f0bf 100644 --- a/slatedb/src/admin.rs +++ b/slatedb/src/admin.rs @@ -1845,6 +1845,7 @@ mod tests { ..Settings::default() }; let write_opts = WriteOptions { + await_durable: false, ..Default::default() }; @@ -1927,7 +1928,10 @@ mod tests { wal_enabled: false, ..Settings::default() }; - let write_opts = WriteOptions::default(); + let write_opts = WriteOptions { + await_durable: false, + ..Default::default() + }; // Two parents with disjoint single-key SSTs (the union path rejects overlaps). for (path, key) in [(&parent_path1, b"a"), (&parent_path2, b"z")] { diff --git a/slatedb/src/batch_write.rs b/slatedb/src/batch_write.rs index 13c2f79731..7a8afe9d43 100644 --- a/slatedb/src/batch_write.rs +++ b/slatedb/src/batch_write.rs @@ -655,6 +655,7 @@ mod tests { #[cfg(dst)] now: 0, seqnum: 42, + ..Default::default() }, ); handler.handle(msg).await.unwrap(); @@ -700,6 +701,7 @@ mod tests { #[cfg(dst)] now: 0, seqnum: 1, + ..Default::default() }, ); handler.handle(msg).await.unwrap(); diff --git a/slatedb/src/cached_object_store/object_store.rs b/slatedb/src/cached_object_store/object_store.rs index d7e6d9a447..1e39852055 100644 --- a/slatedb/src/cached_object_store/object_store.rs +++ b/slatedb/src/cached_object_store/object_store.rs @@ -249,7 +249,7 @@ impl CachedObjectStore { // Second pass: load the selected files in bounded parallelism and cache them. let degree_of_parallelism = 32; - let _result = build_concurrent(files_to_load.into_iter(), degree_of_parallelism, |path| { + let _result = build_concurrent(files_to_load, degree_of_parallelism, |path| { let this = self.clone(); async move { match this @@ -390,7 +390,7 @@ impl CachedObjectStore { // Convert PutPayload to stream and save parts to cache. let entry = self.cache_storage.entry(location, self.part_size_bytes); - let stream = stream::iter(payload.into_iter()).map(Ok::); + let stream = stream::iter(payload).map(Ok::); // Save parts, ignoring errors (cache failures must not fail the PUT). self.save_parts_stream(entry.as_ref(), stream, 0).await.ok(); diff --git a/slatedb/src/checkpoint.rs b/slatedb/src/checkpoint.rs index 21a6dee911..9451675402 100644 --- a/slatedb/src/checkpoint.rs +++ b/slatedb/src/checkpoint.rs @@ -54,7 +54,7 @@ mod tests { use crate::block_cache_policy::BlockCachePolicy; use crate::checkpoint::Checkpoint; use crate::checkpoint::CheckpointCreateResult; - use crate::config::{CheckpointOptions, CheckpointScope, Settings}; + use crate::config::{CheckpointOptions, CheckpointScope, PutOptions, Settings, WriteOptions}; use crate::db::Db; use crate::db_state::{SsTableId, SsTableView}; use crate::format::sst::SsTableFormat; @@ -343,8 +343,16 @@ mod tests { .await .unwrap(); - db.put(b"k1", b"v1").await.unwrap(); - db.put(b"k2", b"v2").await.unwrap(); + let write_options = WriteOptions { + await_durable: false, + ..Default::default() + }; + db.put_with_options(b"k1", b"v1", &PutOptions::default(), &write_options) + .await + .unwrap(); + db.put_with_options(b"k2", b"v2", &PutOptions::default(), &write_options) + .await + .unwrap(); let checkpoint = db .create_checkpoint(CheckpointScope::All, &CheckpointOptions::default()) diff --git a/slatedb/src/clone.rs b/slatedb/src/clone.rs index 3d9127c8bc..4e74af30b6 100644 --- a/slatedb/src/clone.rs +++ b/slatedb/src/clone.rs @@ -1475,6 +1475,7 @@ mod tests { .await .unwrap(); let write_options = WriteOptions { + await_durable: false, ..Default::default() }; let put_options = PutOptions::default(); @@ -1565,6 +1566,7 @@ mod tests { .await .unwrap(); let write_options = WriteOptions { + await_durable: false, ..Default::default() }; let put_options = PutOptions::default(); diff --git a/slatedb/src/compaction_execute_bench.rs b/slatedb/src/compaction_execute_bench.rs index 3ee197b89f..c7d4c71603 100644 --- a/slatedb/src/compaction_execute_bench.rs +++ b/slatedb/src/compaction_execute_bench.rs @@ -182,7 +182,7 @@ impl CompactionExecuteBench { .now() .signed_duration_since(start) .num_milliseconds(); - info!("wrote sst [id={:?}, elapsed_ms={}]", &sst.id, elapsed_ms); + info!("wrote sst [id={:?}, elapsed_ms={}]", sst.id, elapsed_ms); Ok(()) } diff --git a/slatedb/src/compactor.rs b/slatedb/src/compactor.rs index b36c2a2c54..60bd4c344d 100644 --- a/slatedb/src/compactor.rs +++ b/slatedb/src/compactor.rs @@ -671,11 +671,11 @@ impl CompactorEventHandler { 0.0 }; - let percentage = if estimated_source_bytes > 0 { - (compaction.bytes_processed() * 100 / estimated_source_bytes) as u32 - } else { - 0 - }; + let percentage = compaction + .bytes_processed() + .saturating_mul(100) + .checked_div(estimated_source_bytes) + .unwrap_or_default() as u32; debug!( "compaction progress [id={}, progress={}%, processed_bytes={}, estimated_source_bytes={}, elapsed={:.2}s, throughput={}/s]", compaction.id(), @@ -1674,6 +1674,7 @@ mod tests { &v, &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -1686,6 +1687,7 @@ mod tests { &v, &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -2321,6 +2323,7 @@ mod tests { db.delete_with_options( &[b'a'; 16], &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -2434,6 +2437,7 @@ mod tests { db.delete_with_options( &[b'a'; 16], &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -2526,6 +2530,7 @@ mod tests { b"a", &MergeOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -2536,6 +2541,7 @@ mod tests { b"b", &MergeOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -2546,6 +2552,7 @@ mod tests { &vec![b'p'; 128], &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -2558,6 +2565,7 @@ mod tests { b"c", &MergeOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -2568,6 +2576,7 @@ mod tests { b"x", &MergeOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -2578,6 +2587,7 @@ mod tests { &vec![b'p'; 128], &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -2662,6 +2672,7 @@ mod tests { b"a", &MergeOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -2678,6 +2689,7 @@ mod tests { b"b", &MergeOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -2730,6 +2742,7 @@ mod tests { b"a", &MergeOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -2740,6 +2753,7 @@ mod tests { b"b", &MergeOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -2750,6 +2764,7 @@ mod tests { &vec![b'p'; 128], &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -2770,6 +2785,7 @@ mod tests { b"c", &MergeOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -2780,6 +2796,7 @@ mod tests { b"d", &MergeOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -2790,6 +2807,7 @@ mod tests { &vec![b'p'; 128], &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -2865,6 +2883,7 @@ mod tests { b"x", &MergeOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -2875,6 +2894,7 @@ mod tests { &vec![b'p'; 128], &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -2887,6 +2907,7 @@ mod tests { b"y", &MergeOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -2897,6 +2918,7 @@ mod tests { b"z", &MergeOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -2907,6 +2929,7 @@ mod tests { &vec![b'p'; 128], &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -2982,6 +3005,7 @@ mod tests { b"1", &MergeOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -2992,6 +3016,7 @@ mod tests { &vec![b'p'; 128], &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -3004,6 +3029,7 @@ mod tests { b"2", &MergeOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -3014,6 +3040,7 @@ mod tests { &vec![b'p'; 128], &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -3026,6 +3053,7 @@ mod tests { b"3", &MergeOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -3036,6 +3064,7 @@ mod tests { &vec![b'p'; 128], &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -3121,6 +3150,7 @@ mod tests { ttl: Ttl::ExpireAfterMillis(10), }, &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -3137,6 +3167,7 @@ mod tests { &[b'b'; 32], &MergeOptions { ttl: Ttl::NoExpiry }, &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -3208,6 +3239,7 @@ mod tests { b"a", &MergeOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -3218,6 +3250,7 @@ mod tests { b"b", &MergeOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -3228,6 +3261,7 @@ mod tests { &vec![b'p'; 128], &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -3241,6 +3275,7 @@ mod tests { b"new_value", &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -3251,6 +3286,7 @@ mod tests { &vec![b'p'; 128], &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -3312,6 +3348,7 @@ mod tests { ttl: Ttl::ExpireAfterMillis(100), }, &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -3322,6 +3359,7 @@ mod tests { &vec![b'p'; 128], &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -3336,6 +3374,7 @@ mod tests { ttl: Ttl::ExpireAfterMillis(200), }, &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -3346,6 +3385,7 @@ mod tests { &vec![b'p'; 128], &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -3449,6 +3489,7 @@ mod tests { ttl: Ttl::ExpireAtMillis(1000), }, &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -3464,6 +3505,7 @@ mod tests { ttl: Ttl::ExpireAtMillis(1000), }, &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -3564,6 +3606,7 @@ mod tests { ttl: Ttl::ExpireAtMillis(10), }, &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -3579,6 +3622,7 @@ mod tests { ttl: Ttl::ExpireAtMillis(i64::MAX), }, &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -3594,6 +3638,7 @@ mod tests { value, &PutOptions { ttl: Ttl::NoExpiry }, &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -3677,6 +3722,7 @@ mod tests { ttl: Ttl::ExpireAfterMillis(10), }, &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -3690,6 +3736,7 @@ mod tests { value, &PutOptions { ttl: Ttl::Default }, &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -3705,6 +3752,7 @@ mod tests { value, &PutOptions { ttl: Ttl::NoExpiry }, &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -3721,6 +3769,7 @@ mod tests { ttl: Ttl::ExpireAfterMillis(80), }, &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -6110,6 +6159,7 @@ mod tests { value, &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) diff --git a/slatedb/src/compactor_state.rs b/slatedb/src/compactor_state.rs index 4c09ef1df9..ebbc186c64 100644 --- a/slatedb/src/compactor_state.rs +++ b/slatedb/src/compactor_state.rs @@ -801,8 +801,8 @@ impl Compactions { let latest_finished = self .core .recent_compactions - .iter() - .filter_map(|(_, c)| { + .values() + .filter_map(|c| { if c.status().finished() { Some(c.id()) } else { diff --git a/slatedb/src/config.rs b/slatedb/src/config.rs index f9f8b8dffc..c8e92a7fa4 100644 --- a/slatedb/src/config.rs +++ b/slatedb/src/config.rs @@ -527,8 +527,14 @@ impl CloseOptions { /// Configuration for client write operations. `WriteOptions` is supplied for each /// write call and controls the behavior of the write. -#[derive(Clone, Debug, Default)] +#[derive(Clone, Debug)] pub struct WriteOptions { + /// Whether the write call waits for the returned handle to become durable before returning. + /// + /// Defaults to `true` for compatibility with the fork's v0.15 write contract. Callers that + /// want v0.16's non-blocking write behavior can set this to `false` and await the returned + /// [`crate::WriteHandle`] when durability is required. + pub await_durable: bool, #[cfg(dst)] /// Force the current timestamp for DST operations. See #719 for details. pub now: i64, @@ -539,6 +545,17 @@ pub struct WriteOptions { pub seqnum: u64, } +impl Default for WriteOptions { + fn default() -> Self { + Self { + await_durable: true, + #[cfg(dst)] + now: 0, + seqnum: 0, + } + } +} + /// Configuration for client put operations. `PutOptions` is supplied for each /// row inserted. This differs from [`WriteOptions`] in that a write may encompass /// multiple puts (such as the case with batched writes) diff --git a/slatedb/src/db.rs b/slatedb/src/db.rs index ae32de0331..2421aaa857 100644 --- a/slatedb/src/db.rs +++ b/slatedb/src/db.rs @@ -325,7 +325,24 @@ impl DbInner { self.maybe_apply_backpressure().await?; self.write_notifier.send(batch_msg)?; - rx.await? + let write_handle = rx.await??; + if options.await_durable { + let seq = write_handle.seq; + let mut status_subscription = self.status_manager.subscribe(); + let status = status_subscription + .wait_for(|status| status.durable_seq >= seq || status.close_reason.is_some()) + .await + .map_err(|_| SlateDBError::Closed)?; + if status.durable_seq < seq { + self.check_closed()?; + warn!( + "durable seq {} not advanced past write seq {} and db not closed", + status.durable_seq, seq + ); + return Err(SlateDBError::InvalidDBState); + } + } + Ok(write_handle) } #[inline] @@ -1401,10 +1418,8 @@ impl Db { /// Write a value into the database with default `PutOptions` and /// `WriteOptions`. /// - /// This method returns after updating the in-memory WAL and MemTable. It - /// does not wait for the write to become durable in object storage. Call - /// [`WriteHandle::await_durable`] on the returned handle to wait for this - /// write, or [`Db::flush`] to flush all pending writes. + /// The default [`WriteOptions`] wait for the write to become durable in + /// object storage before returning. /// /// ## Arguments /// - `key`: the key to write @@ -1440,10 +1455,10 @@ impl Db { /// Write a value into the database with custom `PutOptions` and `WriteOptions`. /// - /// This method returns after updating the in-memory WAL and MemTable. It - /// does not wait for the write to become durable in object storage. Call - /// [`WriteHandle::await_durable`] on the returned handle to wait for this - /// write, or [`Db::flush`] to flush all pending writes. + /// Durability behavior is controlled by [`WriteOptions::await_durable`]. + /// When it is `false`, call [`WriteHandle::await_durable`] on the returned + /// handle to wait for this write, or [`Db::flush`] to flush all pending + /// writes. /// /// ## Arguments /// - `key`: the key to write @@ -1491,9 +1506,8 @@ impl Db { /// from a prior read, a zero-copy buffer pool, or a client that produces /// [`Bytes`] directly). /// - /// This method does not wait for durability. Call - /// [`WriteHandle::await_durable`] on the returned handle to wait for this - /// write, or [`Db::flush`] to flush all pending writes. + /// The default [`WriteOptions`] wait for the write to become durable in + /// object storage before returning. pub async fn put_bytes(&self, key: Bytes, value: Bytes) -> Result { self.put_bytes_with_options(key, value, &PutOptions::default(), &WriteOptions::default()) .await @@ -1503,9 +1517,10 @@ impl Db { /// `PutOptions` and `WriteOptions`. See [`Db::put_bytes`] for why this /// form exists. /// - /// This method does not wait for durability. Call - /// [`WriteHandle::await_durable`] on the returned handle to wait for this - /// write, or [`Db::flush`] to flush all pending writes. + /// Durability behavior is controlled by [`WriteOptions::await_durable`]. + /// When it is `false`, call [`WriteHandle::await_durable`] on the returned + /// handle to wait for this write, or [`Db::flush`] to flush all pending + /// writes. pub async fn put_bytes_with_options( &self, key: Bytes, @@ -1520,10 +1535,8 @@ impl Db { /// Delete a key from the database with default `WriteOptions`. /// - /// This method returns after updating the in-memory WAL and MemTable. It - /// does not wait for the delete to become durable in object storage. Call - /// [`WriteHandle::await_durable`] on the returned handle to wait for this - /// delete, or [`Db::flush`] to flush all pending writes. + /// The default [`WriteOptions`] wait for the delete to become durable in + /// object storage before returning. /// /// ## Arguments /// - `key`: the key to delete @@ -1554,10 +1567,10 @@ impl Db { /// Delete a key from the database with custom `WriteOptions`. /// - /// This method returns after updating the in-memory WAL and MemTable. It - /// does not wait for the delete to become durable in object storage. Call - /// [`WriteHandle::await_durable`] on the returned handle to wait for this - /// delete, or [`Db::flush`] to flush all pending writes. + /// Durability behavior is controlled by [`WriteOptions::await_durable`]. + /// When it is `false`, call [`WriteHandle::await_durable`] on the returned + /// handle to wait for this delete, or [`Db::flush`] to flush all pending + /// writes. /// /// ## Arguments /// - `key`: the key to delete @@ -1593,10 +1606,8 @@ impl Db { /// Merge a value into the database with default `MergeOptions` and `WriteOptions`. /// - /// This method returns after updating the in-memory WAL and MemTable. It - /// does not wait for the merge to become durable in object storage. Call - /// [`WriteHandle::await_durable`] on the returned handle to wait for this - /// merge, or [`Db::flush`] to flush all pending writes. + /// The default [`WriteOptions`] wait for the merge to become durable in + /// object storage before returning. /// /// Merge operations allow applications to bypass the traditional read/modify/write cycle /// by expressing partial updates using an associative operator. The merge operator must @@ -1654,10 +1665,10 @@ impl Db { /// Merge a value into the database with custom `MergeOptions` and `WriteOptions`. /// - /// This method returns after updating the in-memory WAL and MemTable. It - /// does not wait for the merge to become durable in object storage. Call - /// [`WriteHandle::await_durable`] on the returned handle to wait for this - /// merge, or [`Db::flush`] to flush all pending writes. + /// Durability behavior is controlled by [`WriteOptions::await_durable`]. + /// When it is `false`, call [`WriteHandle::await_durable`] on the returned + /// handle to wait for this merge, or [`Db::flush`] to flush all pending + /// writes. /// /// Merge operations allow applications to bypass the traditional read/modify/write cycle /// by expressing partial updates using an associative operator. The merge operator must @@ -1730,10 +1741,8 @@ impl Db { /// block other gets and writes until the batch is written to the WAL (or memtable if /// WAL is disabled). /// - /// This method returns after updating the in-memory WAL and MemTable. It - /// does not wait for the batch to become durable in object storage. Call - /// [`WriteHandle::await_durable`] on the returned handle to wait for this - /// batch, or [`Db::flush`] to flush all pending writes. + /// The default [`WriteOptions`] wait for the batch to become durable in + /// object storage before returning. /// /// ## Arguments /// - `batch`: the batch of put/delete operations to write @@ -1771,10 +1780,10 @@ impl Db { /// block other gets and writes until the batch is written to the WAL (or memtable if /// WAL is disabled). /// - /// This method returns after updating the in-memory WAL and MemTable. It - /// does not wait for the batch to become durable in object storage. Call - /// [`WriteHandle::await_durable`] on the returned handle to wait for this - /// batch, or [`Db::flush`] to flush all pending writes. + /// Durability behavior is controlled by [`WriteOptions::await_durable`]. + /// When it is `false`, call [`WriteHandle::await_durable`] on the returned + /// handle to wait for this batch, or [`Db::flush`] to flush all pending + /// writes. /// /// ## Arguments /// - `batch`: the batch of put/delete operations to write @@ -1976,7 +1985,7 @@ impl Db { let env_vars = std::env::vars().map(|(key, value)| (key.to_ascii_lowercase(), value)); let (object_store, path) = parse_url_opts(&url, env_vars).map_err(SlateDBError::from)?; if !path.as_ref().is_empty() { - return Err(SlateDBError::InvalidObjectStorePath(path.to_string()))?; + return Err(SlateDBError::InvalidObjectStorePath(path.to_string()).into()); } Ok(Arc::from(object_store)) } @@ -2148,8 +2157,9 @@ impl DbCacheManagerOps for Db { /// Handle returned from write operations, containing metadata about the write. /// -/// Write operations return this handle without waiting for durability. Call -/// [`WriteHandle::await_durable`] to wait until this write is durable in object +/// Write operations return this handle after applying their configured +/// [`WriteOptions`]. When [`WriteOptions::await_durable`] is `false`, call +/// [`WriteHandle::await_durable`] to wait until the write is durable in object /// storage. /// /// This structure is designed to be extensible for future enhancements. @@ -3083,7 +3093,10 @@ mod tests { b"px:b", b"vb_dirty", &PutOptions::default(), - &WriteOptions::default(), + &WriteOptions { + await_durable: false, + ..Default::default() + }, ) .await .unwrap(); @@ -3260,6 +3273,7 @@ mod tests { &value, &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -3310,6 +3324,7 @@ mod tests { value, &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -3362,6 +3377,7 @@ mod tests { b"test_value", &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -3445,6 +3461,7 @@ mod tests { b"test_value", &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -3510,6 +3527,7 @@ mod tests { b"test_value", &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -3567,6 +3585,7 @@ mod tests { b"test_value", &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -3604,7 +3623,17 @@ mod tests { .await .unwrap(); - db.put(b"test_key", b"test_value").await.unwrap(); + db.put_with_options( + b"test_key", + b"test_value", + &PutOptions::default(), + &WriteOptions { + await_durable: false, + ..Default::default() + }, + ) + .await + .unwrap(); assert_eq!( lookup_metric(&metrics_recorder, crate::db_stats::L0_FLUSH_BYTES).unwrap_or(0), @@ -3633,7 +3662,17 @@ mod tests { .await .unwrap(); - db.put(b"test_key", b"test_value").await.unwrap(); + db.put_with_options( + b"test_key", + b"test_value", + &PutOptions::default(), + &WriteOptions { + await_durable: false, + ..Default::default() + }, + ) + .await + .unwrap(); db.close_with_options(CloseOptions::default().with_flush_type(Some(FlushType::Wal))) .await @@ -3677,7 +3716,17 @@ mod tests { .await .unwrap(); - db.put(b"test_key", b"test_value").await.unwrap(); + db.put_with_options( + b"test_key", + b"test_value", + &PutOptions::default(), + &WriteOptions { + await_durable: false, + ..Default::default() + }, + ) + .await + .unwrap(); db.close_with_options(CloseOptions::default().with_flush_type(None)) .await @@ -3710,7 +3759,18 @@ mod tests { .await .unwrap(); - let handle = db.put(b"key", b"value").await.unwrap(); + let handle = db + .put_with_options( + b"key", + b"value", + &PutOptions::default(), + &WriteOptions { + await_durable: false, + ..Default::default() + }, + ) + .await + .unwrap(); db.close_with_options(CloseOptions::default().with_flush_type(None)) .await @@ -3839,6 +3899,7 @@ mod tests { b"world", &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -3886,6 +3947,7 @@ mod tests { .unwrap(); let put_options = PutOptions::default(); let write_options = WriteOptions { + await_durable: false, ..Default::default() }; let get_memory_options = ReadOptions::new().with_durability_filter(Memory); @@ -3945,6 +4007,7 @@ mod tests { ttl: Default::default(), }, &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -4482,6 +4545,7 @@ mod tests { db.delete_with_options( &[b'b'; 4], &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -4533,6 +4597,7 @@ mod tests { .await .unwrap(); let write_options = WriteOptions { + await_durable: false, ..Default::default() }; @@ -4552,6 +4617,7 @@ mod tests { // the memtable will not be flushed to l0, and the test will hang // at this put_with_options call. let write_options = WriteOptions { + await_durable: false, ..Default::default() }; clock.set(10); @@ -4710,6 +4776,7 @@ mod tests { .unwrap(); let write_options: WriteOptions = WriteOptions { + await_durable: false, ..Default::default() }; let put_options = PutOptions::default(); @@ -4838,6 +4905,7 @@ mod tests { value1, &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -4852,6 +4920,7 @@ mod tests { value2, &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -4953,6 +5022,7 @@ mod tests { value, &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -5042,6 +5112,7 @@ mod tests { .await .unwrap(); let write_options = WriteOptions { + await_durable: false, ..Default::default() }; let put_options = PutOptions::default(); @@ -5154,6 +5225,7 @@ mod tests { .unwrap(); let write_options = WriteOptions { + await_durable: false, ..Default::default() }; @@ -5284,6 +5356,7 @@ mod tests { value1, &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -5298,6 +5371,7 @@ mod tests { value2, &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -5371,6 +5445,7 @@ mod tests { value1, &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -5441,6 +5516,7 @@ mod tests { .unwrap(); let metrics_recorder_clone = metrics_recorder.clone(); let write_opts = WriteOptions { + await_durable: false, ..Default::default() }; @@ -5562,6 +5638,7 @@ mod tests { .await .unwrap(); let write_opts = WriteOptions { + await_durable: false, ..Default::default() }; @@ -5715,6 +5792,7 @@ mod tests { // do all flushes manually let write_opts = WriteOptions { + await_durable: false, ..Default::default() }; @@ -5822,6 +5900,7 @@ mod tests { b"val1", &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -5832,6 +5911,7 @@ mod tests { b"val2", &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -5842,6 +5922,7 @@ mod tests { b"val3", &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -5975,6 +6056,7 @@ mod tests { "bar".as_bytes(), &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -6031,6 +6113,7 @@ mod tests { "bla".as_bytes(), &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -6082,6 +6165,7 @@ mod tests { .delete_with_options( "foo".as_bytes(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -6148,6 +6232,7 @@ mod tests { "uncommitted2".as_bytes(), &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -6159,6 +6244,7 @@ mod tests { "uncommitted4".as_bytes(), &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -6411,7 +6497,15 @@ mod tests { fail_parallel::cfg(fp_registry.clone(), "write-wal-sst-io-error", "panic").unwrap(); let result = db - .put(b"foo", b"bar") + .put_with_options( + b"foo", + b"bar", + &PutOptions::default(), + &WriteOptions { + await_durable: false, + ..Default::default() + }, + ) .await .unwrap() .await_durable() @@ -6445,7 +6539,15 @@ mod tests { // Trigger a WAL write, which should not advance the manifest WAL ID let result = db - .put(b"foo", b"bar") + .put_with_options( + b"foo", + b"bar", + &PutOptions::default(), + &WriteOptions { + await_durable: false, + ..Default::default() + }, + ) .await .unwrap() .await_durable() @@ -6507,6 +6609,7 @@ mod tests { b"bar", &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -6550,7 +6653,18 @@ mod tests { .await .unwrap(); - let handle = db.put(b"foo", b"bar").await.unwrap(); + let handle = db + .put_with_options( + b"foo", + b"bar", + &PutOptions::default(), + &WriteOptions { + await_durable: false, + ..Default::default() + }, + ) + .await + .unwrap(); let durability_wait = tokio::spawn(async move { handle.await_durable().await }); tokio::task::yield_now().await; assert!(!durability_wait.is_finished()); @@ -7104,6 +7218,7 @@ mod tests { b"1", &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -7119,6 +7234,7 @@ mod tests { b"1", &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -7151,6 +7267,7 @@ mod tests { b"1", &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -7171,6 +7288,7 @@ mod tests { b"1", &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -7252,6 +7370,7 @@ mod tests { &[b'j'; 8], &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -7263,6 +7382,7 @@ mod tests { &[b'k'; 8], &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -7517,6 +7637,7 @@ mod tests { let value = format!("{}{}", "v".repeat(i), i); let put_option = PutOptions::default(); let write_option = WriteOptions { + await_durable: false, ..Default::default() }; db.put_with_options(key.as_bytes(), value.clone(), &put_option, &write_option) @@ -8084,6 +8205,7 @@ mod tests { // do a write and flush memtable only (not wal) let write_opts = WriteOptions { + await_durable: false, ..Default::default() }; db.put_with_options(&b"foo", &b"bar", &PutOptions::default(), &write_opts) @@ -8314,6 +8436,7 @@ mod tests { b"value1", &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -8483,6 +8606,7 @@ mod tests { value, &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -8502,6 +8626,7 @@ mod tests { b"value2", &put_opts, &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -8516,6 +8641,7 @@ mod tests { .delete_with_options( b"key1", &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -8533,6 +8659,7 @@ mod tests { .write_with_options( batch, &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -8563,6 +8690,7 @@ mod tests { .write_with_options( batch, &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -8580,6 +8708,7 @@ mod tests { .write_with_options( batch, &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -8596,6 +8725,7 @@ mod tests { .write_with_options( batch, &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -8619,6 +8749,7 @@ mod tests { .write_with_options( WriteBatch::new(), &WriteOptions { + await_durable: false, ..Default::default() }, None, @@ -8995,6 +9126,7 @@ mod tests { .await .unwrap(); let write_opts = WriteOptions { + await_durable: false, ..Default::default() }; @@ -9396,6 +9528,7 @@ mod tests { b"value1", &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -9406,6 +9539,7 @@ mod tests { b"value2", &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -9495,6 +9629,7 @@ mod tests { b"aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa0", &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -9510,6 +9645,7 @@ mod tests { val.as_bytes(), &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -9649,6 +9785,7 @@ mod tests { b"base", &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -9666,6 +9803,7 @@ mod tests { operand, &MergeOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -9765,6 +9903,7 @@ mod tests { ttl: Ttl::ExpireAfterMillis(50), }, &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -9797,6 +9936,7 @@ mod tests { ttl: Ttl::ExpireAfterMillis(50), }; let write_opts = WriteOptions { + await_durable: false, ..Default::default() }; @@ -9872,6 +10012,7 @@ mod tests { ttl: Ttl::ExpireAtMillis(500), }, &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -9886,6 +10027,7 @@ mod tests { ttl: Ttl::ExpireAtMillis(500), }, &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -10010,6 +10152,7 @@ mod tests { db.write_with_options( batch, &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -10095,6 +10238,7 @@ mod tests { // when: two writes (the second triggers maybe_apply_backpressure for the first's bytes) let write_opts = WriteOptions { + await_durable: false, ..Default::default() }; db.put_with_options(b"k1", b"v1", &PutOptions::default(), &write_opts) @@ -10135,6 +10279,7 @@ mod tests { b"v1", &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -10146,6 +10291,7 @@ mod tests { b"v2", &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -10185,6 +10331,7 @@ mod tests { b"v1", &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -10474,6 +10621,7 @@ mod tests { .unwrap(); let write_opts = WriteOptions { + await_durable: false, ..Default::default() }; @@ -10633,6 +10781,7 @@ mod tests { &value, &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -10671,6 +10820,7 @@ mod tests { b"value", &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -10698,6 +10848,7 @@ mod tests { .await .unwrap(); let write_opts = WriteOptions { + await_durable: false, ..Default::default() }; for (key, value) in [ @@ -10769,6 +10920,7 @@ mod tests { .await .unwrap(); let write_opts = WriteOptions { + await_durable: false, ..Default::default() }; source @@ -10817,6 +10969,7 @@ mod tests { let mut rx = db.subscribe(); assert!(rx.borrow_and_update().list_segments().is_empty()); let write_opts = WriteOptions { + await_durable: false, ..Default::default() }; @@ -10873,6 +11026,7 @@ mod tests { b"v1", &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -10940,6 +11094,7 @@ mod tests { b"v1", &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -11249,6 +11404,7 @@ mod tests { .await .unwrap(); let write_opts = WriteOptions { + await_durable: false, ..Default::default() }; for (k, v) in [(b"aaa-1".as_slice(), b"v1"), (b"bbb-1".as_slice(), b"v2")] { diff --git a/slatedb/src/db_cache_manager.rs b/slatedb/src/db_cache_manager.rs index 773274b4e1..e9929ba629 100644 --- a/slatedb/src/db_cache_manager.rs +++ b/slatedb/src/db_cache_manager.rs @@ -270,6 +270,7 @@ mod tests { &value, &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) diff --git a/slatedb/src/db_reader.rs b/slatedb/src/db_reader.rs index 6023a8a809..1765ff73c0 100644 --- a/slatedb/src/db_reader.rs +++ b/slatedb/src/db_reader.rs @@ -2382,6 +2382,7 @@ mod tests { .await .unwrap(); let write_opts = WriteOptions { + await_durable: false, ..Default::default() }; db.put_with_options(b"abc-1", b"v1", &PutOptions::default(), &write_opts) @@ -2475,6 +2476,7 @@ mod tests { .await .unwrap(); let write_opts = WriteOptions { + await_durable: false, ..Default::default() }; db.put_with_options(b"abc-1", b"v1", &PutOptions::default(), &write_opts) @@ -3061,6 +3063,25 @@ mod tests { ); } + #[tokio::test] + async fn default_durable_write_is_visible_to_reader_opened_immediately() { + let object_store: Arc = Arc::new(InMemory::new()); + let path = Path::from("/tmp/default_durable_write_is_visible_to_reader"); + let test_provider = TestProvider::new(path, Arc::clone(&object_store)); + let db = test_provider.new_db(Settings::default()).await.unwrap(); + + db.put(b"key", b"value").await.unwrap(); + + let reader = test_provider + .new_db_reader(DbReaderOptions::default(), None, None) + .await + .unwrap(); + assert_eq!( + reader.get(b"key").await.unwrap(), + Some(Bytes::from_static(b"value")) + ); + } + #[tokio::test] async fn reader_snapshot_should_remain_stable_while_wal_replay_advances() { let object_store: Arc = Arc::new(InMemory::new()); @@ -3076,7 +3097,10 @@ mod tests { .await .unwrap(); - let write_options = WriteOptions::default(); + let write_options = WriteOptions { + await_durable: false, + ..Default::default() + }; db.put_with_options(b"key", b"v1", &PutOptions::default(), &write_options) .await .unwrap(); @@ -3160,7 +3184,10 @@ mod tests { .build() .await .unwrap(); - let write_options = WriteOptions::default(); + let write_options = WriteOptions { + await_durable: false, + ..Default::default() + }; db.put_with_options(b"key", b"v1", &PutOptions::default(), &write_options) .await .unwrap(); @@ -3269,7 +3296,10 @@ mod tests { .build() .await .unwrap(); - let write_options = WriteOptions::default(); + let write_options = WriteOptions { + await_durable: false, + ..Default::default() + }; db.put_with_options(b"key", b"v1", &PutOptions::default(), &write_options) .await .unwrap(); @@ -3426,7 +3456,10 @@ mod tests { }) .await .unwrap(); - let write_options = WriteOptions::default(); + let write_options = WriteOptions { + await_durable: false, + ..Default::default() + }; db.put_with_options(b"key", b"v1", &PutOptions::default(), &write_options) .await .unwrap(); @@ -3527,7 +3560,10 @@ mod tests { }) .await .unwrap(); - let write_options = WriteOptions::default(); + let write_options = WriteOptions { + await_durable: false, + ..Default::default() + }; db.put_with_options(b"a", b"old", &PutOptions::default(), &write_options) .await .unwrap(); @@ -5089,7 +5125,10 @@ mod tests { b"key", b"value", &PutOptions::default(), - &WriteOptions::default(), + &WriteOptions { + await_durable: false, + ..Default::default() + }, ) .await .unwrap(); @@ -5293,6 +5332,7 @@ mod tests { b"a", &MergeOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -5303,6 +5343,7 @@ mod tests { b"b", &MergeOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -5345,6 +5386,7 @@ mod tests { b"c", &MergeOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -5355,6 +5397,7 @@ mod tests { b"d", &MergeOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) diff --git a/slatedb/src/db_snapshot.rs b/slatedb/src/db_snapshot.rs index d6dbee3ebe..c8a6b3b49b 100644 --- a/slatedb/src/db_snapshot.rs +++ b/slatedb/src/db_snapshot.rs @@ -897,6 +897,7 @@ mod tests { b"value2", &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) diff --git a/slatedb/src/db_state.rs b/slatedb/src/db_state.rs index 13bc49915f..deebde0501 100644 --- a/slatedb/src/db_state.rs +++ b/slatedb/src/db_state.rs @@ -496,7 +496,7 @@ pub struct SortedRun { /// Held behind an `Arc` so cloning a `SortedRun` (e.g. per read in the /// scan path) is a single refcount bump rather than a deep clone of every /// view's `Bytes` handles. - sst_views: Arc<[SsTableView]>, + pub sst_views: Arc<[SsTableView]>, } impl SortedRun { diff --git a/slatedb/src/db_transaction.rs b/slatedb/src/db_transaction.rs index 354524805d..2864ca311a 100644 --- a/slatedb/src/db_transaction.rs +++ b/slatedb/src/db_transaction.rs @@ -802,10 +802,8 @@ impl DbTransaction { /// write-write conflicts) is deferred to the task that processes the /// WriteBatch, which ensures the atomicity of transactions. /// - /// A successful commit does not wait for durability in object storage. - /// Call [`WriteHandle::await_durable`] on the returned handle when the - /// result is `Some`, or call [`crate::Db::flush`] to flush all pending - /// writes. + /// The default write options wait for a successful commit to become + /// durable in object storage. /// /// If the transaction's write batch is empty, this operation is a no-op and returns `Ok(())` /// immediately without any database interaction. Since it's impossible to have read-write @@ -828,10 +826,10 @@ impl DbTransaction { /// This method behaves the same as [`DbTransaction::commit`], but allows callers /// to specify custom [`WriteOptions`]. /// - /// A successful commit does not wait for durability in object storage. - /// Call [`WriteHandle::await_durable`] on the returned handle when the - /// result is `Some`, or call [`crate::Db::flush`] to flush all pending - /// writes. + /// Durability behavior is controlled by [`WriteOptions::await_durable`]. + /// When it is `false`, call [`WriteHandle::await_durable`] on the returned + /// handle when the result is `Some`, or call [`crate::Db::flush`] to flush + /// all pending writes. /// /// ## Arguments /// - `options`: the write options to use for the commit @@ -1709,9 +1707,12 @@ mod tests { txn.put(b"k", b"v").unwrap(); // Commits return without waiting for durability. - txn.commit_with_options(&WriteOptions::default()) - .await - .unwrap(); + txn.commit_with_options(&WriteOptions { + await_durable: false, + ..Default::default() + }) + .await + .unwrap(); // Memory (in-memory) read should see the value let val = db @@ -2789,6 +2790,7 @@ mod tests { txn.put(b"key1", b"value1").unwrap(); let handle = txn .commit_with_options(&WriteOptions { + await_durable: false, ..Default::default() }) .await @@ -2806,6 +2808,7 @@ mod tests { txn.put_with_options(b"key2", b"value2", &put_opts).unwrap(); let handle = txn .commit_with_options(&WriteOptions { + await_durable: false, ..Default::default() }) .await @@ -2820,6 +2823,7 @@ mod tests { txn.delete(b"key1").unwrap(); let handle = txn .commit_with_options(&WriteOptions { + await_durable: false, ..Default::default() }) .await @@ -2839,6 +2843,7 @@ mod tests { let txn = db.begin(IsolationLevel::Snapshot).await.unwrap(); let result = txn .commit_with_options(&WriteOptions { + await_durable: false, ..Default::default() }) .await diff --git a/slatedb/src/format/block.rs b/slatedb/src/format/block.rs index f9b22a1f61..c0688f2b47 100644 --- a/slatedb/src/format/block.rs +++ b/slatedb/src/format/block.rs @@ -86,7 +86,9 @@ fn compute_prefix(lhs: &[u8], rhs: &[u8]) -> usize { } fn compute_prefix_chunks(lhs: &[u8], rhs: &[u8]) -> usize { - let off = std::iter::zip(lhs.chunks_exact(N), rhs.chunks_exact(N)) + let (lhs_chunks, _) = lhs.as_chunks::(); + let (rhs_chunks, _) = rhs.as_chunks::(); + let off = std::iter::zip(lhs_chunks, rhs_chunks) .take_while(|(a, b)| a == b) .count() * N; diff --git a/slatedb/src/format/block_v2.rs b/slatedb/src/format/block_v2.rs index 1d3c6851af..5ae595acf6 100644 --- a/slatedb/src/format/block_v2.rs +++ b/slatedb/src/format/block_v2.rs @@ -64,7 +64,9 @@ fn compute_prefix(lhs: &[u8], rhs: &[u8]) -> usize { /// the overhead of chunk iteration and the benefits of bulk comparison. fn compute_prefix_chunks(lhs: &[u8], rhs: &[u8]) -> usize { // Compare N-byte chunks until we find one that differs - let off = std::iter::zip(lhs.chunks_exact(N), rhs.chunks_exact(N)) + let (lhs_chunks, _) = lhs.as_chunks::(); + let (rhs_chunks, _) = rhs.as_chunks::(); + let off = std::iter::zip(lhs_chunks, rhs_chunks) .take_while(|(a, b)| a == b) .count() * N; diff --git a/slatedb/src/mem_table.rs b/slatedb/src/mem_table.rs index f29a1d17b2..b132cf465f 100644 --- a/slatedb/src/mem_table.rs +++ b/slatedb/src/mem_table.rs @@ -849,11 +849,9 @@ mod tests { let sample_table = sample::table(runner.rng(), 500, 10); let kv_table = WritableKVTable::new(); - let mut seq = 1; - for (key, value) in &sample_table { + for (seq, (key, value)) in (1..).zip(&sample_table) { let row_entry = RowEntry::new_value(key, value, seq); kv_table.put(row_entry); - seq += 1; } runner diff --git a/slatedb/src/memtable_flusher/manifest_writer.rs b/slatedb/src/memtable_flusher/manifest_writer.rs index a58aebc255..f138c07a8b 100644 --- a/slatedb/src/memtable_flusher/manifest_writer.rs +++ b/slatedb/src/memtable_flusher/manifest_writer.rs @@ -731,10 +731,7 @@ impl ManifestWriterHandler { self.db.oracle.advance_durable_seq(uploaded.last_seq); } self.resolve_pending_flushes(); - for (checkpoint, result) in attached_checkpoints - .into_iter() - .zip(checkpoint_results.into_iter()) - { + for (checkpoint, result) in attached_checkpoints.into_iter().zip(checkpoint_results) { debug!("checkpoint created [id={}]", result.id); let _ = checkpoint.sender.send(Ok(result)); } diff --git a/slatedb/src/ops.rs b/slatedb/src/ops.rs index 4a9bb6d6ac..cade7baff0 100644 --- a/slatedb/src/ops.rs +++ b/slatedb/src/ops.rs @@ -243,10 +243,10 @@ pub trait DbReadOps { /// allowing consumers to write generic code or test doubles over the writer /// surface without depending on the concrete `Db` type. /// -/// Write methods return after updating the in-memory WAL and MemTable. They do -/// not wait for the write to become durable in object storage. Call -/// [`WriteHandle::await_durable`] on the returned handle to wait for one write, -/// or [`Self::flush`] to flush all pending writes. +/// Durability behavior is controlled by [`WriteOptions::await_durable`], which +/// defaults to `true`. When it is `false`, call [`WriteHandle::await_durable`] +/// on the returned handle to wait for one write, or [`Self::flush`] to flush +/// all pending writes. #[async_trait::async_trait] pub trait DbWriteOps { /// The transaction type returned by [`Self::begin`]. Stub @@ -257,7 +257,7 @@ pub trait DbWriteOps { /// Write a value into the database with default `PutOptions` and /// `WriteOptions`. /// - /// This method does not wait for durability. See [`DbWriteOps`] for + /// The default write options wait for durability. See [`DbWriteOps`] for /// details. /// /// ## Arguments @@ -278,8 +278,7 @@ pub trait DbWriteOps { /// Write a value into the database with custom `PutOptions` and /// `WriteOptions`. /// - /// This method does not wait for durability. See [`DbWriteOps`] for - /// details. + /// Durability behavior follows `write_opts`. See [`DbWriteOps`] for details. /// /// ## Arguments /// - `key`: the key to write @@ -302,7 +301,7 @@ pub trait DbWriteOps { /// Delete a key from the database with default `WriteOptions`. /// - /// This method does not wait for durability. See [`DbWriteOps`] for + /// The default write options wait for durability. See [`DbWriteOps`] for /// details. /// /// ## Arguments @@ -317,8 +316,7 @@ pub trait DbWriteOps { /// Delete a key from the database with custom `WriteOptions`. /// - /// This method does not wait for durability. See [`DbWriteOps`] for - /// details. + /// Durability behavior follows `options`. See [`DbWriteOps`] for details. /// /// ## Arguments /// - `key`: the key to delete @@ -335,7 +333,7 @@ pub trait DbWriteOps { /// Merge a value into the database with default `MergeOptions` and /// `WriteOptions`. /// - /// This method does not wait for durability. See [`DbWriteOps`] for + /// The default write options wait for durability. See [`DbWriteOps`] for /// details. /// /// Merge operations allow applications to bypass the traditional @@ -367,8 +365,7 @@ pub trait DbWriteOps { /// Merge a value into the database with custom `MergeOptions` and /// `WriteOptions`. /// - /// This method does not wait for durability. See [`DbWriteOps`] for - /// details. + /// Durability behavior follows `write_opts`. See [`DbWriteOps`] for details. /// /// ## Arguments /// - `key`: the key to merge into @@ -392,7 +389,7 @@ pub trait DbWriteOps { /// Write a batch of put/delete operations atomically to the database. /// - /// This method does not wait for durability. See [`DbWriteOps`] for + /// The default write options wait for durability. See [`DbWriteOps`] for /// details. /// /// ## Arguments @@ -408,8 +405,7 @@ pub trait DbWriteOps { /// Write a batch of put/delete operations atomically to the database with /// custom `WriteOptions`. /// - /// This method does not wait for durability. See [`DbWriteOps`] for - /// details. + /// Durability behavior follows `options`. See [`DbWriteOps`] for details. /// /// ## Arguments /// - `batch`: the batch of operations to write diff --git a/slatedb/src/paths.rs b/slatedb/src/paths.rs index 2bd157b9d9..376b00d263 100644 --- a/slatedb/src/paths.rs +++ b/slatedb/src/paths.rs @@ -62,11 +62,11 @@ impl PathResolver { } pub(crate) fn wal_path(&self) -> Path { - Path::from(format!("{}/{}/", &self.root_path, WAL_PATH)) + Path::from(format!("{}/{}/", self.root_path, WAL_PATH)) } pub(crate) fn compacted_path(&self) -> Path { - Path::from(format!("{}/{}/", &self.root_path, COMPACTED_PATH)) + Path::from(format!("{}/{}/", self.root_path, COMPACTED_PATH)) } pub(crate) fn parse_table_id(&self, path: &Path) -> Result, SlateDBError> { diff --git a/slatedb/src/segment_iterator.rs b/slatedb/src/segment_iterator.rs index 6ee26c8a2f..2929a1c12f 100644 --- a/slatedb/src/segment_iterator.rs +++ b/slatedb/src/segment_iterator.rs @@ -393,7 +393,7 @@ async fn build_sr_range_iters( let table_store = ctx.table_store.clone(); let opts = ctx.sst_iter_options.clone(); let stats = ctx.db_stats.clone(); - build_concurrent(overlapping.into_iter(), ctx.max_parallel, move |sr| { + build_concurrent(overlapping, ctx.max_parallel, move |sr| { let table_store = table_store.clone(); let range = range.clone(); let opts = opts.clone(); diff --git a/slatedb/src/test_utils.rs b/slatedb/src/test_utils.rs index d730abfd71..07407909c1 100644 --- a/slatedb/src/test_utils.rs +++ b/slatedb/src/test_utils.rs @@ -370,20 +370,14 @@ pub(crate) async fn seed_database( wait_for_durability: bool, ) -> Result<(), crate::Error> { let put_options = PutOptions::default(); - let write_options = WriteOptions::default(); - let mut last_handle = None; + let write_options = WriteOptions { + await_durable: wait_for_durability, + ..Default::default() + }; for (key, value) in table.iter() { - last_handle = Some( - db.put_with_options(key, value, &put_options, &write_options) - .await?, - ); - } - - if wait_for_durability { - if let Some(handle) = last_handle { - handle.await_durable().await?; - } + db.put_with_options(key, value, &put_options, &write_options) + .await?; } Ok(()) diff --git a/slatedb/src/transaction_manager.rs b/slatedb/src/transaction_manager.rs index 70ef591ae0..3a3dc0cb89 100644 --- a/slatedb/src/transaction_manager.rs +++ b/slatedb/src/transaction_manager.rs @@ -1584,7 +1584,7 @@ mod tests { // For every active transaction, verify the smoking-gun invariant: // rw_conflict <=> exists culprit in recent_committed_txns that satisfies rules let inner = txn_manager.inner.read(); - for (_id, txn) in inner.active_txns.iter() { + for txn in inner.active_txns.values() { // Only meaningful for SSI; read-only transactions can also have reads, so include them. let rw_conflict = inner.has_read_write_conflict( &txn.read_keys, diff --git a/slatedb/src/wal/slatedb/iterator.rs b/slatedb/src/wal/slatedb/iterator.rs index c643e9cb59..289cecec68 100644 --- a/slatedb/src/wal/slatedb/iterator.rs +++ b/slatedb/src/wal/slatedb/iterator.rs @@ -411,7 +411,7 @@ impl WalIteratorTrait for SlateDbWalIterator { [wal_id={}, min_seq={}, last_seq={}]", rows.last_consumed_wal_file_id, min_seq, last_seq, ); - error!("{}", &msg); + error!("{}", msg); let error = Arc::from(Box::::from(msg)); return self.terminate(Err(WalError::InternalError(error))); diff --git a/slatedb/tests/db.rs b/slatedb/tests/db.rs index 86d3608da9..2d82e717bd 100644 --- a/slatedb/tests/db.rs +++ b/slatedb/tests/db.rs @@ -57,6 +57,7 @@ async fn test_replay_wal_then_write() { value.as_bytes(), &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -89,6 +90,7 @@ async fn test_replay_wal_then_write() { b"new_value", &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) @@ -242,6 +244,7 @@ async fn test_concurrent_writers_and_readers() { i.to_be_bytes().as_ref(), &PutOptions::default(), &WriteOptions { + await_durable: false, ..Default::default() }, ) diff --git a/slatedb/tests/prefix_filter.rs b/slatedb/tests/prefix_filter.rs index 5f8eae0861..99510a4353 100644 --- a/slatedb/tests/prefix_filter.rs +++ b/slatedb/tests/prefix_filter.rs @@ -119,7 +119,10 @@ mod composite_filters { async fn write_sample_data(db: &Db) { let put = PutOptions::default(); - let write = WriteOptions::default(); + let write = WriteOptions { + await_durable: false, + ..WriteOptions::default() + }; // Write each batch in its own SST so multiple SSTs participate in the // read path and the filter has something to actually skip. for batch in [SAMPLE_USERS, SAMPLE_NON_USERS] { @@ -296,7 +299,10 @@ mod subrange { .expect("failed to build db"); let put = PutOptions::default(); - let write = WriteOptions::default(); + let write = WriteOptions { + await_durable: false, + ..WriteOptions::default() + }; let ssts: &[&[&[u8]]] = &[ &[b"aaa1", b"ccc1"], // sandwich &[b"bbb1", b"bbb2", b"bbb3", b"bbb4"], @@ -487,7 +493,10 @@ mod empty_prefix_filter { let db = open_db(store.clone(), recorder.clone()).await; let put = PutOptions::default(); - let write = WriteOptions::default(); + let write = WriteOptions { + await_durable: false, + ..WriteOptions::default() + }; for key in [b"a".as_slice(), b"b".as_slice()] { db.put_with_options(key, b"v", &put, &write) .await @@ -574,7 +583,10 @@ mod prop_test { async fn write_keys(db: &Db, keys: &[Vec]) { let put_opts = PutOptions::default(); - let write_opts = WriteOptions::default(); + let write_opts = WriteOptions { + await_durable: false, + ..Default::default() + }; for (i, key) in keys.iter().enumerate() { let value = format!("v{}", i).into_bytes(); db.put_with_options(key, &value, &put_opts, &write_opts) From 41ce04071f51d51a85774af834fc77947cafd34d Mon Sep 17 00:00:00 2001 From: Matthew Sanetra <41018997+matthewsanetra@users.noreply.github.com> Date: Wed, 2 Sep 2026 10:05:55 +0100 Subject: [PATCH 65/65] fix reader options in generated bindings --- bindings/go/uniffi/slatedb.go | 5 +++++ bindings/go/uniffi/slatedb_test.go | 4 ++++ .../src/test/java/io/slatedb/uniffi/TestSupport.java | 2 +- bindings/node/tests/support.mjs | 1 + 4 files changed, 11 insertions(+), 1 deletion(-) diff --git a/bindings/go/uniffi/slatedb.go b/bindings/go/uniffi/slatedb.go index 00e0e49a38..af4e20de97 100644 --- a/bindings/go/uniffi/slatedb.go +++ b/bindings/go/uniffi/slatedb.go @@ -10410,6 +10410,8 @@ func (_ FfiDestroyerReadOptions) Destroy(value ReadOptions) { type ReaderOptions struct { // How often the reader polls for new manifests and WAL data, in milliseconds. ManifestPollIntervalMs uint64 + // How frequently an open reader probes the exact next WAL ID. + WalPollIntervalMs uint64 // Lifetime of an internally managed checkpoint, in milliseconds. CheckpointLifetimeMs uint64 // Maximum size of one in-memory table used while replaying WAL data. @@ -10425,6 +10427,7 @@ type ReaderOptions struct { func (r *ReaderOptions) Destroy() { FfiDestroyerUint64{}.Destroy(r.ManifestPollIntervalMs) + FfiDestroyerUint64{}.Destroy(r.WalPollIntervalMs) FfiDestroyerUint64{}.Destroy(r.CheckpointLifetimeMs) FfiDestroyerUint64{}.Destroy(r.MaxMemtableBytes) FfiDestroyerBool{}.Destroy(r.SkipWalReplay) @@ -10444,6 +10447,7 @@ func (c FfiConverterReaderOptions) Read(reader io.Reader) ReaderOptions { FfiConverterUint64INSTANCE.Read(reader), FfiConverterUint64INSTANCE.Read(reader), FfiConverterUint64INSTANCE.Read(reader), + FfiConverterUint64INSTANCE.Read(reader), FfiConverterBoolINSTANCE.Read(reader), FfiConverterOptionalUint32INSTANCE.Read(reader), } @@ -10459,6 +10463,7 @@ func (c FfiConverterReaderOptions) LowerExternal(value ReaderOptions) ExternalCR func (c FfiConverterReaderOptions) Write(writer io.Writer, value ReaderOptions) { FfiConverterUint64INSTANCE.Write(writer, value.ManifestPollIntervalMs) + FfiConverterUint64INSTANCE.Write(writer, value.WalPollIntervalMs) FfiConverterUint64INSTANCE.Write(writer, value.CheckpointLifetimeMs) FfiConverterUint64INSTANCE.Write(writer, value.MaxMemtableBytes) FfiConverterBoolINSTANCE.Write(writer, value.SkipWalReplay) diff --git a/bindings/go/uniffi/slatedb_test.go b/bindings/go/uniffi/slatedb_test.go index 4e677dab43..95d199bcf6 100644 --- a/bindings/go/uniffi/slatedb_test.go +++ b/bindings/go/uniffi/slatedb_test.go @@ -1456,6 +1456,7 @@ func TestDbReaderRefreshBehavior(t *testing.T) { t.Helper() if err := builder.WithOptions(slatedb.ReaderOptions{ ManifestPollIntervalMs: 100, + WalPollIntervalMs: 100, CheckpointLifetimeMs: 1000, MaxMemtableBytes: 64 * 1024 * 1024, SkipWalReplay: false, @@ -1488,6 +1489,7 @@ func TestDbReaderWalReplayBehavior(t *testing.T) { t.Helper() if err := builder.WithOptions(slatedb.ReaderOptions{ ManifestPollIntervalMs: 100, + WalPollIntervalMs: 100, CheckpointLifetimeMs: 1000, MaxMemtableBytes: 64 * 1024 * 1024, SkipWalReplay: false, @@ -1534,6 +1536,7 @@ func TestDbReaderWalReplayBehavior(t *testing.T) { t.Helper() if err := builder.WithOptions(slatedb.ReaderOptions{ ManifestPollIntervalMs: 100, + WalPollIntervalMs: 100, CheckpointLifetimeMs: 1000, MaxMemtableBytes: 64 * 1024 * 1024, SkipWalReplay: true, @@ -1566,6 +1569,7 @@ func TestDbReaderWalReplayBehavior(t *testing.T) { t.Helper() if err := builder.WithOptions(slatedb.ReaderOptions{ ManifestPollIntervalMs: 100, + WalPollIntervalMs: 100, CheckpointLifetimeMs: 1000, MaxMemtableBytes: 64 * 1024 * 1024, SkipWalReplay: true, diff --git a/bindings/java/slatedb-uniffi/src/test/java/io/slatedb/uniffi/TestSupport.java b/bindings/java/slatedb-uniffi/src/test/java/io/slatedb/uniffi/TestSupport.java index b0b336794b..48049306df 100644 --- a/bindings/java/slatedb-uniffi/src/test/java/io/slatedb/uniffi/TestSupport.java +++ b/bindings/java/slatedb-uniffi/src/test/java/io/slatedb/uniffi/TestSupport.java @@ -362,7 +362,7 @@ static ScanOptions scanOptions(long readAheadBytes, boolean cacheBlocks, long ma } static ReaderOptions readerOptions(boolean skipWalReplay) { - return new ReaderOptions(100L, 1000L, 64L * 1024 * 1024, skipWalReplay, null); + return new ReaderOptions(100L, 100L, 1000L, 64L * 1024 * 1024, skipWalReplay, null); } private static Throwable unwrap(Throwable thrown) { diff --git a/bindings/node/tests/support.mjs b/bindings/node/tests/support.mjs index 4d91b0197b..cfba301ad3 100644 --- a/bindings/node/tests/support.mjs +++ b/bindings/node/tests/support.mjs @@ -63,6 +63,7 @@ export function scanOptions(readAheadBytes, cacheBlocks, maxFetchTasks) { export function readerOptions(skipWalReplay) { return { manifest_poll_interval_ms: 100, + wal_poll_interval_ms: 100, checkpoint_lifetime_ms: 1_000, max_memtable_bytes: 64 * 1024 * 1024, skip_wal_replay: skipWalReplay,