diff --git a/.github/workflows/nightly.yaml b/.github/workflows/nightly.yaml index b3b509d808..24639cfb13 100644 --- a/.github/workflows/nightly.yaml +++ b/.github/workflows/nightly.yaml @@ -15,7 +15,7 @@ permissions: jobs: # Run and save nightly microbenchmark data so PRs have a fresh baseline to compare against microbenchmarks: - runs-on: ubuntu-latest + runs-on: warp-ubuntu-latest-arm64-8x steps: - uses: actions/checkout@v4 @@ -54,159 +54,8 @@ jobs: if: steps.update_microbenchmark_result.outcome == 'failure' run: exit 1 - benchmarks: - runs-on: warp-ubuntu-latest-x64-16x - timeout-minutes: 20 - steps: - - uses: actions/checkout@v4 - - - name: Install dependencies - run: | - sudo apt-get update - sudo apt-get install -y traceroute - sudo snap install aws-cli --classic - - - name: System information - env: - AWS_ACCESS_KEY_ID: ${{ secrets.TIGRIS_AWS_ACCESS_KEY_ID }} - AWS_SECRET_ACCESS_KEY: ${{ secrets.TIGRIS_AWS_SECRET_ACCESS_KEY }} - AWS_BUCKET: ${{ secrets.TIGRIS_AWS_BUCKET_BENCHER }} - AWS_REGION: auto - AWS_ENDPOINT: https://t3.storage.dev - run: | - echo "=== CPU ===" - lscpu - echo -e "\n=== Memory ===" - free -h - echo -e "\n=== Disk Space ===" - df -h - echo -e "\n=== Workspace Directory ===" - du -sh ${{ github.workspace }} - echo -e "\n=== Network ===" - traceroute t3.storage.dev - echo -e "Generating 1 gig file" - dd if=/dev/urandom of=/tmp/1gig bs=1G count=1 - echo -e "Uploading 1 gig file" - time aws s3 cp --endpoint-url $AWS_ENDPOINT /tmp/1gig s3://${{ secrets.TIGRIS_AWS_BUCKET_BENCHER }}/1gig - echo -e "Downloading 1 gig file" - time aws s3 cp --endpoint-url $AWS_ENDPOINT s3://${{ secrets.TIGRIS_AWS_BUCKET_BENCHER }}/1gig /tmp/1gig - echo -e "Deleting 1 gig file" - time aws s3 rm --endpoint-url $AWS_ENDPOINT s3://${{ secrets.TIGRIS_AWS_BUCKET_BENCHER }}/1gig - - - name: Download previous mermaid plots - uses: actions/cache/restore@v4 - with: - path: ./target/bencher/results/mermaid - key: ${{ runner.os }}-mermaid-plots-${{ github.run_id }} - restore-keys: | - ${{ runner.os }}-mermaid-plots- - - - name: Run benchmark - env: - CLOUD_PROVIDER: aws - AWS_ACCESS_KEY_ID: ${{ secrets.TIGRIS_AWS_ACCESS_KEY_ID }} - AWS_SECRET_ACCESS_KEY: ${{ secrets.TIGRIS_AWS_SECRET_ACCESS_KEY }} - AWS_BUCKET: ${{ secrets.TIGRIS_AWS_BUCKET_BENCHER }} - AWS_REGION: auto - AWS_ENDPOINT: https://t3.storage.dev - SLATEDB_BENCH_CLEAN: true - RUST_LOG: info - run: | - aws s3 rm --endpoint-url $AWS_ENDPOINT s3://${{ secrets.TIGRIS_AWS_BUCKET_BENCHER }} --recursive - ./slatedb-bencher/benchmark-db.sh - - - name: Save mermaid plots cache - uses: actions/cache/save@v4 - with: - path: ./target/bencher/results/mermaid - key: ${{ runner.os }}-mermaid-plots-${{ github.run_id }} - - - name: Add mermaid diagrams to summary - run: | - echo "# SlateDB Benchmark Results" >> $GITHUB_STEP_SUMMARY - echo "" >> $GITHUB_STEP_SUMMARY - - # Add each mermaid diagram to the summary - for mermaid_file in target/bencher/results/mermaid/*.mermaid; do - if [ -f "$mermaid_file" ]; then - echo "" >> $GITHUB_STEP_SUMMARY - echo '```mermaid' >> $GITHUB_STEP_SUMMARY - cat "$mermaid_file" >> $GITHUB_STEP_SUMMARY - echo '```' >> $GITHUB_STEP_SUMMARY - echo "" >> $GITHUB_STEP_SUMMARY - fi - done - - echo "Mermaid diagrams added to GitHub Actions summary!" - echo "Total diagrams: $(ls -1 target/bencher/results/mermaid/*.mermaid 2>/dev/null | wc -l)" - - transaction-benchmarks: - runs-on: warp-ubuntu-latest-x64-16x - # Must run after `benchmarks` because both jobs clear the same bucket prefix; - # running concurrently can delete each other's manifest/object files mid-run. - # This also reduces noise between the two tests since the bucket's resources - # (network, disk, etc) should be used by only one test at a time, giving more - # stable results. - needs: benchmarks - timeout-minutes: 30 - steps: - - uses: actions/checkout@v4 - - - name: Install dependencies - run: | - sudo apt-get update - sudo apt-get install -y traceroute - sudo snap install aws-cli --classic - - - name: Download previous transaction mermaid plots - uses: actions/cache/restore@v4 - with: - path: ./target/bencher/transaction-results/mermaid - key: ${{ runner.os }}-txn-mermaid-plots-${{ github.run_id }} - restore-keys: | - ${{ runner.os }}-txn-mermaid-plots- - - - name: Run transaction benchmark - env: - CLOUD_PROVIDER: aws - AWS_ACCESS_KEY_ID: ${{ secrets.TIGRIS_AWS_ACCESS_KEY_ID }} - AWS_SECRET_ACCESS_KEY: ${{ secrets.TIGRIS_AWS_SECRET_ACCESS_KEY }} - AWS_BUCKET: ${{ secrets.TIGRIS_AWS_BUCKET_BENCHER }} - AWS_REGION: auto - AWS_ENDPOINT: https://t3.storage.dev - SLATEDB_BENCH_CLEAN: true - RUST_LOG: info - run: | - aws s3 rm --endpoint-url $AWS_ENDPOINT s3://${{ secrets.TIGRIS_AWS_BUCKET_BENCHER }} --recursive - ./slatedb-bencher/benchmark-transaction.sh - - - name: Save transaction mermaid plots cache - uses: actions/cache/save@v4 - with: - path: ./target/bencher/transaction-results/mermaid - key: ${{ runner.os }}-txn-mermaid-plots-${{ github.run_id }} - - - name: Add transaction mermaid diagrams to summary - run: | - echo "# SlateDB Transaction Benchmark Results" >> $GITHUB_STEP_SUMMARY - echo "" >> $GITHUB_STEP_SUMMARY - - # Add each mermaid diagram to the summary - for mermaid_file in target/bencher/transaction-results/mermaid/*.mermaid; do - if [ -f "$mermaid_file" ]; then - echo "" >> $GITHUB_STEP_SUMMARY - echo '```mermaid' >> $GITHUB_STEP_SUMMARY - cat "$mermaid_file" >> $GITHUB_STEP_SUMMARY - echo '```' >> $GITHUB_STEP_SUMMARY - echo "" >> $GITHUB_STEP_SUMMARY - fi - done - - echo "Transaction mermaid diagrams added to GitHub Actions summary!" - echo "Total diagrams: $(ls -1 target/bencher/transaction-results/mermaid/*.mermaid 2>/dev/null | wc -l)" - microbenchmark-pprofs: - runs-on: ubuntu-latest + runs-on: warp-ubuntu-latest-arm64-8x steps: - uses: actions/checkout@v4 @@ -234,10 +83,15 @@ jobs: # install instructions from https://github.com/polarsignals/pprofme/tree/main - name: Install pprofme run: | - curl -LO https://github.com/polarsignals/pprofme/releases/latest/download/pprofme_$(uname)_$(uname -m) + PPROFME_ARCH=$(uname -m) + if [ "$PPROFME_ARCH" = "aarch64" ]; then + PPROFME_ARCH=arm64 + fi + PPROFME_BINARY="pprofme_$(uname)_${PPROFME_ARCH}" + curl -fLO "https://github.com/polarsignals/pprofme/releases/latest/download/${PPROFME_BINARY}" curl -sL https://github.com/polarsignals/pprofme/releases/latest/download/pprofme_checksums.txt | shasum --ignore-missing -a 256 --check - chmod a+x pprofme_$(uname)_$(uname -m) - sudo mv pprofme_$(uname)_$(uname -m) /usr/local/bin/pprofme + chmod a+x "${PPROFME_BINARY}" + sudo mv "${PPROFME_BINARY}" /usr/local/bin/pprofme - name: Upload pprofs and add links to summary run: | diff --git a/.github/workflows/release.yaml b/.github/workflows/release.yaml index 6eec0a7e3e..96765b5c65 100644 --- a/.github/workflows/release.yaml +++ b/.github/workflows/release.yaml @@ -32,6 +32,18 @@ jobs: with: ssh-key: ${{ secrets.RELEASE_SSH_KEY }} ref: ${{ github.ref }} + fetch-depth: 0 + + # Select the most recent stable release tag reachable from the release branch. + - name: Find previous release tag + id: previous_release + run: | + previous_tag="$(git describe \ + --tags \ + --abbrev=0 \ + --match 'v[0-9]*.[0-9]*.[0-9]*' \ + --exclude '*-*')" + echo "tag=${previous_tag}" >> "${GITHUB_OUTPUT}" # Set up Rust stable for installing cargo-edit - name: Setup Rust stable @@ -67,6 +79,7 @@ jobs: tag: v${{ github.event.inputs.version }} name: v${{ github.event.inputs.version }} generateReleaseNotes: true + generateReleaseNotesPreviousTag: ${{ steps.previous_release.outputs.tag }} token: ${{ secrets.GITHUB_TOKEN }} # Publish crate chain to crates.io in dependency order diff --git a/Cargo.lock b/Cargo.lock index fcaea23145..8cb07d77fa 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -32,9 +32,9 @@ dependencies = [ [[package]] name = "aho-corasick" -version = "1.1.4" +version = "1.1.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ddd31a130427c27518df266943a5308ed92d4b226cc639f5a8f1002816174301" +checksum = "c982642fa9e8606056828ee9a8505737230110bb1099153c79efe865c59d12ba" dependencies = [ "memchr", ] @@ -127,15 +127,15 @@ dependencies = [ [[package]] name = "anyhow" -version = "1.0.102" +version = "1.0.104" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7f202df86484c868dbad7eaa557ef785d5c66295e41b460ef922eca0723b842c" +checksum = "330a5ed07fa54e4702c9d6c4174f74427fc0ef6e214bbd677ae50a5099946470" [[package]] name = "arrayvec" -version = "0.7.6" +version = "0.7.8" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7c02d123df017efcdfbd739ef81735b36c5ba83ec3c59c80a9d7ecc718f92e50" +checksum = "d3fb67a6e08acf24fdeccbac2cb6ac4305825bd1f117462e0e6f2f193345ad56" [[package]] name = "askama" @@ -164,7 +164,7 @@ dependencies = [ "rustc-hash", "serde", "serde_derive", - "syn", + "syn 2.0.119", ] [[package]] @@ -217,13 +217,13 @@ dependencies = [ [[package]] name = "async-trait" -version = "0.1.89" +version = "0.1.91" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9035ad2d096bed7955a320ee7e2230574d28fd3c3a0f186cbea1ff3c7eed5dbb" +checksum = "ae36dc4177970ef04fde5178d3e2429882def40e57a451f919c098f72baa6cec" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 3.0.4", ] [[package]] @@ -243,15 +243,15 @@ checksum = "1505bd5d3d116872e7271a6d4e16d81d0c8570876c8de68093a09ac269d8aac0" [[package]] name = "autocfg" -version = "1.5.0" +version = "1.5.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c08606f8c3cbf4ce6ec8e28fb0014a2c086708fe954eaa885384a6165172e7e8" +checksum = "f2032f911046de80f0a198e0901378627c33f59ea0ac00e363d481118bd70a53" [[package]] name = "aws-lc-rs" -version = "1.17.0" +version = "1.17.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5ec2f1fc3ec205783a5da9a7e6c1509cc69dedf09a1949e412c1e18469326d00" +checksum = "00bdb5da18dac48ca2cc7cd4a98e533e8635a58e2361d13a1a4ee3888e0d72f1" dependencies = [ "aws-lc-sys", "zeroize", @@ -259,14 +259,15 @@ dependencies = [ [[package]] name = "aws-lc-sys" -version = "0.41.0" +version = "0.43.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1a2f9779ce85b93ab6170dd940ad0169b5766ff848247aff13bb788b832fe3f4" +checksum = "43103168cc76fe62678a375e722fc9cb3a0146159ac5828bc4f0dfd755c2224c" dependencies = [ "cc", "cmake", "dunce", "fs_extra", + "pkg-config", ] [[package]] @@ -342,9 +343,9 @@ checksum = "bef38d45163c2f1dde094a7dfd33ccf595c92905c8f8f4fdc18d06fb1037718a" [[package]] name = "bitflags" -version = "2.11.1" +version = "2.13.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c4512299f36f043ab09a583e57bceb5a5aab7a73db1805848e8fef3c9e8c78b3" +checksum = "b588b76d00fde79687d7646a9b5bdf3cc0f655e0bbd080335a95d7e96f3587da" [[package]] name = "block-buffer" @@ -357,30 +358,30 @@ dependencies = [ [[package]] name = "bumpalo" -version = "3.20.2" +version = "3.20.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5d20789868f4b01b2f2caec9f5c4e0213b41e3e5702a50157d699ae31ced2fcb" +checksum = "72f5acc6cb2ba439de613abc23857ec3d78374d8ed5ac84e9d11336e87da8649" [[package]] name = "bytemuck" -version = "1.25.0" +version = "1.25.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c8efb64bd706a16a1bdde310ae86b351e4d21550d98d056f22f8a7f7a2183fec" +checksum = "95832e849adfb21180ccb6826a99da14e5d266ae5c2e668e1602cf234f153797" [[package]] name = "bytes" -version = "1.11.1" +version = "1.12.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1e748733b7cbc798e1434b6ac524f0c1ff2ab456fe201501e6497c8417a4fc33" +checksum = "fc652a48c352aef3ea3aed32080501cf3ef6ed5da78602a020c991775b0aff04" dependencies = [ "serde", ] [[package]] name = "camino" -version = "1.2.2" +version = "1.2.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e629a66d692cb9ff1a1c664e41771b3dcaf961985a9774c0eb0bd1b51cf60a48" +checksum = "bb1307f12aa967b5a58416e87b3653360e0fd614a016b6e970db08fecbb1b80d" dependencies = [ "serde_core", ] @@ -405,7 +406,7 @@ dependencies = [ "semver", "serde", "serde_json", - "thiserror 2.0.18", + "thiserror 2.0.20", ] [[package]] @@ -416,9 +417,9 @@ checksum = "37b2a672a2cb129a2e41c10b1224bb368f9f37a2b16b612598138befd7b37eb5" [[package]] name = "cc" -version = "1.2.62" +version = "1.4.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a1dce859f0832a7d088c4f1119888ab94ef4b5d6795d1ce05afb7fe159d79f98" +checksum = "5add81bb678e6cb321aff7fa0dc7689ad82b112dbc032cea19f91d6b8e3582b9" dependencies = [ "find-msvc-tools", "jobserver", @@ -434,15 +435,15 @@ checksum = "9330f8b2ff13f34540b44e946ef35111825727b38d33286ef986142615121801" [[package]] name = "cfg_aliases" -version = "0.2.1" +version = "0.2.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "613afe47fcd5fac7ccf1db93babcb082c5994d996f20b8b159f2ad1658eb5724" +checksum = "f079e83a288787bcd14a6aea84cee5c87a67c5a3e660c30f557a3d24761b3527" [[package]] name = "chacha20" -version = "0.10.0" +version = "0.10.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6f8d983286843e49675a4b7a2d174efe136dc93a18d69130dd18198a6c167601" +checksum = "d524456ba66e72eb8b115ff89e01e497f8e6d11d78b70b1aa13c0fbd97540a81" dependencies = [ "cfg-if", "cpufeatures", @@ -451,9 +452,9 @@ dependencies = [ [[package]] name = "chrono" -version = "0.4.44" +version = "0.4.45" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c673075a2e0e5f4a1dde27ce9dee1ea4558c7ffe648f576438a20ca1d2acc4b0" +checksum = "1aa79e62e7697b8e29b513a68abacf485adcd1fe8284a4316c5ae868e6633327" dependencies = [ "iana-time-zone", "js-sys", @@ -492,9 +493,9 @@ dependencies = [ [[package]] name = "clap" -version = "4.6.1" +version = "4.6.6" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1ddb117e43bbf7dacf0a4190fef4d345b9bad68dfc649cb349e7d17d28428e51" +checksum = "473c7e07f409a8d772161724aa8db6a765a2532a70f9667eeb7b49d3d02fbdca" dependencies = [ "clap_builder", "clap_derive", @@ -502,9 +503,9 @@ dependencies = [ [[package]] name = "clap_builder" -version = "4.6.0" +version = "4.6.6" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "714a53001bf66416adb0e2ef5ac857140e7dc3a0c48fb28b2f10762fc4b5069f" +checksum = "7b48fea5a88e9ae728a2dcbedbfc0e730f7d60da42e1cb049a83c9fb8b789889" dependencies = [ "anstream", "anstyle", @@ -514,14 +515,14 @@ dependencies = [ [[package]] name = "clap_derive" -version = "4.6.1" +version = "4.6.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f2ce8604710f6733aa641a2b3731eaa1e8b3d9973d5e3565da11800813f997a9" +checksum = "d012d2b9d65aca7f18f4d9878a045bc17899bba951561ba5ec3c2ba1eed9a061" dependencies = [ "heck 0.5.0", "proc-macro2", "quote", - "syn", + "syn 3.0.4", ] [[package]] @@ -572,9 +573,9 @@ dependencies = [ [[package]] name = "console" -version = "0.16.3" +version = "0.16.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d64e8af5551369d19cf50138de61f1c42074ab970f74e99be916646777f8fc87" +checksum = "4fe5f465a4f6fee88fad41b85d990f84c835335e85b5d9e6e63e0d06d28cba7c" dependencies = [ "encode_unicode", "libc", @@ -685,18 +686,18 @@ dependencies = [ [[package]] name = "crossbeam-channel" -version = "0.5.15" +version = "0.5.16" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "82b8f8f868b36967f9606790d1903570de9ceaf870a7bf9fbbd3016d636a2cb2" +checksum = "d85363c37faeca707aef026efa9f3b34d077bce547e48f770770625c6013679e" dependencies = [ "crossbeam-utils", ] [[package]] name = "crossbeam-deque" -version = "0.8.6" +version = "0.8.7" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9dd111b7b7f7d55b72c0a6ae361660ee5853c9af73f70c3c2ef6858b950e2e51" +checksum = "5181e0de7b61eb03a81e347d6dd8797bae9da5146707b51077e2d71a54ec0ceb" dependencies = [ "crossbeam-epoch", "crossbeam-utils", @@ -704,9 +705,9 @@ dependencies = [ [[package]] name = "crossbeam-epoch" -version = "0.9.18" +version = "0.9.20" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5b82ac4a3c2ca9c3460964f020e1402edd5753411d7737aa39c3714ad1b5420e" +checksum = "2d6914041f254d6e9176c01941b21115dcfb7089e55135a35411081bd106ef3f" dependencies = [ "crossbeam-utils", ] @@ -723,9 +724,9 @@ dependencies = [ [[package]] name = "crossbeam-utils" -version = "0.8.21" +version = "0.8.22" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d0a5c400df2834b80a4c3327b3aad3a4c4cd4de0629063962b03235697506a28" +checksum = "61803da095bee82a81bb1a452ecc25d3b2f1416d1897eb86430c6159ef717c17" [[package]] name = "crunchy" @@ -782,9 +783,6 @@ name = "deranged" version = "0.5.8" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "7cd812cc2bc1d69d4764bd80df88b4317eaef9e773c75226407d9bc0876b211c" -dependencies = [ - "powerfmt", -] [[package]] name = "digest" @@ -807,13 +805,13 @@ dependencies = [ [[package]] name = "displaydoc" -version = "0.2.5" +version = "0.2.7" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "97369cbbc041bc366949bc74d34658d6cda5621039731c6310521892a3a20ae0" +checksum = "c6232dd377dcc64799954cbd3a9bb882e9cdc1308ccd87b1c098f1fb2eaf82a8" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 3.0.4", ] [[package]] @@ -851,16 +849,16 @@ checksum = "f88959de2d447fd3eddcf1909d1f19fe084e27a056a6904203dc5d8b9e771c1e" dependencies = [ "rust_decimal", "serde", - "thiserror 2.0.18", + "thiserror 2.0.20", "time", "winnow 0.6.26", ] [[package]] name = "either" -version = "1.15.0" +version = "1.18.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "48c757948c5ede0e46177b7add2e67155f70e33c07fea8284df6576da70b3719" +checksum = "252afb9ae5eaa683babdc6a068b3f5726eb19e05070c731f9b2a23a7c3e8ed34" [[package]] name = "encode_unicode" @@ -885,7 +883,7 @@ checksum = "44f23cf4b44bfce11a86ace86f8a73ffdec849c9fd00a386a53d278bd9e81fb3" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] @@ -906,11 +904,10 @@ dependencies = [ [[package]] name = "event-listener" -version = "5.4.1" +version = "5.4.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e13b66accf52311f30a0db42147dadea9850cb48cd070028831ae5f5d4b856ab" +checksum = "5a23add41df1562121a9393cb065eab5146a1242410f23a644851e90cfd669d2" dependencies = [ - "concurrent-queue", "parking", "pin-project-lite", ] @@ -927,7 +924,7 @@ dependencies = [ [[package]] name = "examples" -version = "0.15.0" +version = "0.16.0" dependencies = [ "anyhow", "object_store", @@ -944,7 +941,7 @@ checksum = "c29b33a0187823f1fa88b36980227dc96c7504ede2288e7d2a77d9d6d88b260c" dependencies = [ "log", "once_cell", - "rand 0.9.4", + "rand 0.9.5", "tokio", ] @@ -960,9 +957,9 @@ dependencies = [ [[package]] name = "fastrand" -version = "2.4.1" +version = "2.5.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9f1f227452a390804cdb637b74a86990f2a7d7ba4b7d5693aac9b4dd6defd8d6" +checksum = "da7c62ceae207dd37ea5b845da6a0696c799f85e97da1ab5b7910be3c1c80223" [[package]] name = "figment" @@ -1022,7 +1019,7 @@ version = "25.12.19" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "35f6839d7b3b98adde531effaf34f0c2badc6f4735d26fe74709d8e513a96ef3" dependencies = [ - "bitflags 2.11.1", + "bitflags 2.13.1", "rustc_version", ] @@ -1042,12 +1039,6 @@ version = "1.0.7" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "3f9eec918d3f24069decb9af1554cad7c880e2da24a9afd88aca000531ab82c1" -[[package]] -name = "foldhash" -version = "0.1.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d9c4f5dac5e15c24eb999c26181a6ca40b39fe946cbe4c263c7209467bc83af2" - [[package]] name = "foldhash" version = "0.2.0" @@ -1117,7 +1108,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "db907f40a527ca2aa2f40a5f68b32ea58aa70f050cd233518e9ffd402cfba6ce" dependencies = [ "anyhow", - "bitflags 2.11.1", + "bitflags 2.13.1", "cmsketch", "equivalent", "foyer-common", @@ -1161,7 +1152,7 @@ dependencies = [ "mea", "parking_lot", "pin-project", - "rand 0.9.4", + "rand 0.9.5", "serde", "tracing", "twox-hash", @@ -1204,9 +1195,9 @@ checksum = "42703706b716c37f96a77aea830392ad231f44c9e9a67872fa5548707e11b11c" [[package]] name = "futures" -version = "0.3.32" +version = "0.3.33" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8b147ee9d1f6d097cef9ce628cd2ee62288d963e16fb287bd9286455b241382d" +checksum = "a88cf1f829d945f548cf8fec32c61b1f202b6d93b45848602fc02af4b12ad218" dependencies = [ "futures-channel", "futures-core", @@ -1219,9 +1210,9 @@ dependencies = [ [[package]] name = "futures-channel" -version = "0.3.32" +version = "0.3.33" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "07bbe89c50d7a535e539b8c17bc0b49bdb77747034daa8087407d655f3f7cc1d" +checksum = "262590f4fe6afeb0bc83be1daa64e52657fe185690a958af7f3ad0e92085c5ae" dependencies = [ "futures-core", "futures-sink", @@ -1229,15 +1220,15 @@ dependencies = [ [[package]] name = "futures-core" -version = "0.3.32" +version = "0.3.33" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7e3450815272ef58cec6d564423f6e755e25379b217b0bc688e295ba24df6b1d" +checksum = "2cd50c473c80f6d7c3670a752354b8e569b1a7cbfdc0419ec88e5edad85e0dc7" [[package]] name = "futures-executor" -version = "0.3.32" +version = "0.3.33" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "baf29c38818342a3b26b5b923639e7b1f4a61fc5e76102d4b1981c6dc7a7579d" +checksum = "6754879cc9f2c66f88c6e5c35344bb0bdb0708b0352b1201815667c7eabc7458" dependencies = [ "futures-core", "futures-task", @@ -1246,44 +1237,44 @@ dependencies = [ [[package]] name = "futures-io" -version = "0.3.32" +version = "0.3.33" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cecba35d7ad927e23624b22ad55235f2239cfa44fd10428eecbeba6d6a717718" +checksum = "4577ecaa3c4f96589d473f679a71b596316f6641bc350038b962a5daf0085d7a" [[package]] name = "futures-macro" -version = "0.3.32" +version = "0.3.33" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e835b70203e41293343137df5c0664546da5745f82ec9b84d40be8336958447b" +checksum = "2d6d3cde68c518367be28956066ddfef33813991b77a55005a69dae04bf3b10b" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] name = "futures-sink" -version = "0.3.32" +version = "0.3.33" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c39754e157331b013978ec91992bde1ac089843443c49cbc7f46150b0fad0893" +checksum = "e34418ac499d6305c2fb5ad0ed2f6ac998c5f8ca209b4510f7f94242c647e307" [[package]] name = "futures-task" -version = "0.3.32" +version = "0.3.33" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "037711b3d59c33004d3856fbdc83b99d4ff37a24768fa1be9ce3538a1cde4393" +checksum = "b231ed28831efb4a61a08580c4bc233ec56bc009f4cd8f52da2c3cb97df0c109" [[package]] name = "futures-timer" -version = "3.0.3" +version = "3.0.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f288b0a4f20f9a56b5d1da57e2227c661b7b16168e2f72365f57b63326e29b24" +checksum = "af43fadb8a98512d547e37b4e92e0ced13e205c061b87b4623eff01d918d6968" [[package]] name = "futures-util" -version = "0.3.32" +version = "0.3.33" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "389ca41296e6190b48053de0321d02a77f32f8a5d2461dd38762c0593805c6d6" +checksum = "a77a90a256fce34da66415271e30f94ee91c57b04b8a2c042d9cf3220179deaa" dependencies = [ "futures-channel", "futures-core", @@ -1326,25 +1317,23 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "899def5c37c4fd7b2664648c28120ecec138e4d395b459e5ca34f9cce2dd77fd" dependencies = [ "cfg-if", - "js-sys", "libc", "r-efi 5.3.0", "wasip2", - "wasm-bindgen", ] [[package]] name = "getrandom" -version = "0.4.2" +version = "0.4.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0de51e6874e94e7bf76d726fc5d13ba782deca734ff60d5bb2fb2607c7406555" +checksum = "300e883d756b2e4ec94e02791f39b04b522276138852cfc41d9fb7e904106099" dependencies = [ "cfg-if", + "js-sys", "libc", "r-efi 6.0.0", "rand_core 0.10.1", - "wasip2", - "wasip3", + "wasm-bindgen", ] [[package]] @@ -1355,9 +1344,9 @@ checksum = "e629b9b98ef3dd8afe6ca2bd0f89306cec16d43d907889945bc5d6687f2f13c7" [[package]] name = "glob" -version = "0.3.3" +version = "0.3.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0cc23270f6e1808e30a928bdc84dea0b9b4136a8bc82338574f23baf47bbd280" +checksum = "e4eba85ea1d0a966a983acd07deee566e67395d2d96b6fb39e62b5a833f1eb0b" [[package]] name = "gloo-timers" @@ -1384,9 +1373,9 @@ dependencies = [ [[package]] name = "h2" -version = "0.4.14" +version = "0.4.15" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "171fefbc92fe4a4de27e0698d6a5b392d6a0e333506bc49133760b3bcf948733" +checksum = "6cb093c84e8bd9b188d4c4a8cb6579fc016968d14c99882163cd3ff402a4f155" dependencies = [ "atomic-waker", "bytes", @@ -1417,9 +1406,6 @@ name = "hashbrown" version = "0.15.5" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "9229cfe53dfd69f0609a49f65461bd93001ea1ef889cd5529dd176593f5338a1" -dependencies = [ - "foldhash 0.1.5", -] [[package]] name = "hashbrown" @@ -1429,7 +1415,7 @@ checksum = "841d1cc9bed7f9236f321df977030373f4a4163ae1a7dbfe1a51a2c1a51d9100" dependencies = [ "allocator-api2", "equivalent", - "foldhash 0.2.0", + "foldhash", ] [[package]] @@ -1440,7 +1426,7 @@ checksum = "ed5909b6e89a2db4456e54cd5f673791d7eca6732202bbf2a9cc504fe2f9b84a" dependencies = [ "allocator-api2", "equivalent", - "foldhash 0.2.0", + "foldhash", ] [[package]] @@ -1463,9 +1449,9 @@ checksum = "fc0fef456e4baa96da950455cd02c081ca953b141298e41db3fc7e36b1da849c" [[package]] name = "http" -version = "1.4.0" +version = "1.5.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e3ba2a386d7f85a81f119ad7498ebe444d2e22c2af0b86b069416ace48b3311a" +checksum = "918d3568bebf352712bc2ef3d46a8bcf1a75b373be6539de198e9105cbbf9ce0" dependencies = [ "bytes", "itoa", @@ -1473,9 +1459,9 @@ dependencies = [ [[package]] name = "http-body" -version = "1.0.1" +version = "1.1.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1efedce1fb8e6913f23e0c92de8e62cd5b772a67e7b3946df930a62566c93184" +checksum = "ca2a8f2913ee65f60facd6a5905613afaa448497a0230cc41ce022d93290bc2c" dependencies = [ "bytes", "http", @@ -1483,9 +1469,9 @@ dependencies = [ [[package]] name = "http-body-util" -version = "0.1.3" +version = "0.1.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b021d93e26becf5dc7e1b75b1bed1fd93124b374ceb73f43d4d4eafec896a64a" +checksum = "e9f41fd6a08e4d4ec69df65976da761afd5ad5e58a9d4acb46bd1c953a9e3ff2" dependencies = [ "bytes", "futures-core", @@ -1502,24 +1488,24 @@ checksum = "6dbf3de79e51f3d586ab4cb9d5c3e2c14aa28ed23d180cf89b4df0454a69cc87" [[package]] name = "humantime" -version = "2.3.0" +version = "2.4.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "135b12329e5e3ce057a9f972339ea52bc954fe1e9358ef27f95e89716fbc5424" +checksum = "15cdd26707701c53297e2fa6afb323d55fbc1d0810c3aec078ae3ef0424c3c15" [[package]] name = "hybrid-array" -version = "0.4.12" +version = "0.4.14" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9155a582abd142abc056962c29e3ce5ff2ad5469f4246b537ed42c5deba857da" +checksum = "707114b52a152fa7bdb290cd7cd5912d9467273b6d74e21b8d81aca1f8533f6b" dependencies = [ "typenum", ] [[package]] name = "hyper" -version = "1.9.0" +version = "1.11.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6299f016b246a94207e63da54dbe807655bf9e00044f73ded42c3ac5305fbcca" +checksum = "d22053281f852e11534f5198498373cbb59295120a20771d90f7ed1897490a72" dependencies = [ "atomic-waker", "bytes", @@ -1586,7 +1572,7 @@ dependencies = [ "js-sys", "log", "wasm-bindgen", - "windows-core", + "windows-core 0.62.2", ] [[package]] @@ -1680,12 +1666,6 @@ dependencies = [ "zerovec", ] -[[package]] -name = "id-arena" -version = "2.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3d3067d79b975e8844ca9eb072e16b31c3c1c36928edf9c6789548c524d0d954" - [[package]] name = "idna" version = "1.1.0" @@ -1727,9 +1707,9 @@ checksum = "c8fae54786f62fb2918dcfae3d568594e50eb9b5c25bf04371af6fe7516452fb" [[package]] name = "insta" -version = "1.47.2" +version = "1.48.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7b4a6248eb93a4401ed2f37dfe8ea592d3cf05b7cf4f8efa867b6895af7e094e" +checksum = "86f0f8fee8c926415c58d6ae43a08523a26faccb2323f5e6b644fe7dd4ef6b82" dependencies = [ "console", "once_cell", @@ -1739,20 +1719,20 @@ dependencies = [ [[package]] name = "io-uring" -version = "0.7.12" +version = "0.7.13" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4d09b98f7eace8982db770e4408e7470b028ce513ac28fecdc6bf4c30fe92b62" +checksum = "9080b15e63775b9a2ac7dca720f7050a8b955e092ea0f6020a4a80f69998cdc0" dependencies = [ - "bitflags 2.11.1", + "bitflags 2.13.1", "cfg-if", "libc", ] [[package]] name = "ipnet" -version = "2.12.0" +version = "2.12.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d98f6fed1fde3f8c21bc40a1abb88dd75e67924f9cffc3ef95607bad8017f8e2" +checksum = "6a756c3fac73139e83f14c2d742155dd2b78d3ee56597b419a0579b7bdd6dd78" [[package]] name = "is-terminal" @@ -1816,7 +1796,7 @@ dependencies = [ "jni-sys", "log", "simd_cesu8", - "thiserror 2.0.18", + "thiserror 2.0.20", "walkdir", "windows-link 0.2.1", ] @@ -1831,7 +1811,7 @@ dependencies = [ "quote", "rustc_version", "simd_cesu8", - "syn", + "syn 2.0.119", ] [[package]] @@ -1850,28 +1830,27 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "38c0b942f458fe50cdac086d2f946512305e5631e720728f2a61aabcd47a6264" dependencies = [ "quote", - "syn", + "syn 2.0.119", ] [[package]] name = "jobserver" -version = "0.1.34" +version = "0.1.35" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9afb3de4395d6b3e67a780b6de64b51c978ecf11cb9a462c66be7d4ca9039d33" +checksum = "1c00acbd29eabad4a2392fa0e921c874934dbbf4194312ad20f04a0ed67a3cb3" dependencies = [ - "getrandom 0.3.4", + "getrandom 0.4.3", "libc", ] [[package]] name = "js-sys" -version = "0.3.98" +version = "0.3.103" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "67df7112613f8bfd9150013a0314e196f4800d3201ae742489d999db2f979f08" +checksum = "53b44bfcdb3f8d5837a46dae1ca9660a837176eee74a28b229bc626816589102" dependencies = [ "cfg-if", "futures-util", - "once_cell", "wasm-bindgen", ] @@ -1881,17 +1860,11 @@ version = "1.5.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "bbd2bcb4c963f2ddae06a2efc7e9f3591312473c50c6685e1f298068316e66fe" -[[package]] -name = "leb128fmt" -version = "0.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "09edd9e8b54e49e587e4f6295a7d29c3ea94d469cb40ab8ca70b288248a81db2" - [[package]] name = "libc" -version = "0.2.186" +version = "0.2.189" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "68ab91017fe16c622486840e4c83c9a37afeff978bd239b5293d61ece587de66" +checksum = "3eaf3ede3fee6db1a4c2ee091bf8a8b4dccdc6d17f656fb07896ee72867612f2" [[package]] name = "linux-raw-sys" @@ -1916,15 +1889,15 @@ dependencies = [ [[package]] name = "log" -version = "0.4.29" +version = "0.4.33" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5e5032e24019045c762d3c0f28f5b6b8bbf38563a65908389bf7978758920897" +checksum = "0ceec5bc11778974d1bcb055b18002eba7f4b3518b6a0081b3af5f21666da9ad" [[package]] name = "lru" -version = "0.18.0" +version = "0.18.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8a860605968fce16869fd239cf4237a82f3ac470723415db603b0e8b6c8d4fb9" +checksum = "5d2f2f9b4ba7e6b24d95e7e899329d35be83bcded72c8540cdd5368932d1d90a" dependencies = [ "hashbrown 0.17.1", ] @@ -1984,24 +1957,24 @@ dependencies = [ [[package]] name = "mea" -version = "0.6.3" +version = "0.6.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6747f54621d156e1b47eb6b25f39a941b9fc347f98f67d25d8881ff99e8ed832" +checksum = "31fc7d159de0085ab6dd7ff145a9819442cfd3d098f783263120503c3f3e58b0" dependencies = [ "slab", ] [[package]] name = "memchr" -version = "2.8.0" +version = "2.8.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f8ca58f447f06ed17d5fc4043ce1b10dd205e060fb3ce5b979b8ed8e59ff3f79" +checksum = "cf8baf1c55e62ffcace7a9f06f4bd9cd3f0c4beb022d3b367256b91b87513d98" [[package]] name = "memmap2" -version = "0.9.10" +version = "0.9.11" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "714098028fe011992e1c3962653c96b2d578c4b4bce9036e15ff220319b1e0e3" +checksum = "d1219ed1b7f229ee7104d281dd01d6802fe28bb6e95d292942c4daacdeb798c0" dependencies = [ "libc", ] @@ -2033,9 +2006,9 @@ dependencies = [ [[package]] name = "mio" -version = "1.2.0" +version = "1.2.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "50b7e5b27aa02a74bac8c3f23f448f8d87ff11f92d3aac1a6ed369ee08cc56c1" +checksum = "30d65c71f1ce40ab09135ce117d742b9f8a19ff91a41a8b57ed50bc2de59c427" dependencies = [ "libc", "wasi", @@ -2044,11 +2017,11 @@ dependencies = [ [[package]] name = "mixtrics" -version = "0.2.3" +version = "0.2.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fb252c728b9d77c6ef9103f0c81524fa0a3d3b161d0a936295d7fbeff6e04c11" +checksum = "2c46b5adfb7a3ae4996d327a5bdc90e78fec025806dd312bdbe6f07a755e0ec9" dependencies = [ - "itertools 0.14.0", + "itertools 0.15.0", "parking_lot", ] @@ -2089,7 +2062,7 @@ version = "0.31.3" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "cf20d2fde8ff38632c426f1165ed7436270b44f199fc55284c38276f9db47c3d" dependencies = [ - "bitflags 2.11.1", + "bitflags 2.13.1", "cfg-if", "cfg_aliases", "libc", @@ -2154,7 +2127,7 @@ version = "0.3.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "2a180dd8642fa45cdb7dd721cd4c11b1cadd4929ce112ebd8b9f5803cc79d536" dependencies = [ - "bitflags 2.11.1", + "bitflags 2.13.1", ] [[package]] @@ -2178,9 +2151,9 @@ dependencies = [ [[package]] name = "object_store" -version = "0.14.0" +version = "0.14.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "765784b4390c6bcf80316e5a22f4e3661b639c9d8c83246856643c27d8ce9dbe" +checksum = "d354792e39fa5f0009e47623cf8b15b099bf9a652fa55c6f817fe28ac84fea50" dependencies = [ "async-trait", "aws-lc-rs", @@ -2203,13 +2176,13 @@ dependencies = [ "parking_lot", "percent-encoding", "quick-xml", - "rand 0.10.1", + "rand 0.10.2", "reqwest", "rustls-pki-types", "serde", "serde_json", "serde_urlencoded", - "thiserror 2.0.18", + "thiserror 2.0.20", "tokio", "tracing", "url", @@ -2264,7 +2237,7 @@ dependencies = [ "proc-macro2", "proc-macro2-diagnostics", "quote", - "syn", + "syn 2.0.119", ] [[package]] @@ -2324,7 +2297,7 @@ dependencies = [ "proc-macro2", "proc-macro2-diagnostics", "quote", - "syn", + "syn 2.0.119", ] [[package]] @@ -2360,7 +2333,7 @@ checksum = "c96395f0a926bc13b1c17622aaddda1ecb55d49c8f1bf9777e4d877800a43f8b" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] @@ -2411,9 +2384,9 @@ dependencies = [ [[package]] name = "portable-atomic" -version = "1.13.1" +version = "1.15.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c33a9471896f1c69cecef8d20cbe2f7accd12527ce60845ff44c153bb2a21b49" +checksum = "05c8b63e8d9609db387f0324918f81d68fe27748f084ef092fb35954d0539a85" [[package]] name = "potential_utf" @@ -2463,16 +2436,6 @@ dependencies = [ "zerocopy", ] -[[package]] -name = "prettyplease" -version = "0.2.37" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "479ca8adacdd7ce8f1fb39ce9ecccbfe93a3f1344b3d0d97f20bc0196208f62b" -dependencies = [ - "proc-macro2", - "syn", -] - [[package]] name = "proc-macro-crate" version = "3.5.0" @@ -2484,9 +2447,9 @@ dependencies = [ [[package]] name = "proc-macro2" -version = "1.0.106" +version = "1.0.107" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8fd00f0bb2e90d81d1044c2b32617f68fcb9fa3bb7640c23e9c748e53fb30934" +checksum = "985e7ec9bb745e6ce6535b544d84d6cd6f7ad8bd711c398938ae983b91a766d9" dependencies = [ "unicode-ident", ] @@ -2499,7 +2462,7 @@ checksum = "af066a9c399a26e020ada66a034357a868728e72cd426f3adcd35f80d88d88c8" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.119", "version_check", "yansi", ] @@ -2512,9 +2475,9 @@ checksum = "4b45fcc2344c680f5025fe57779faef368840d0bd1f42f216291f0dc4ace4744" dependencies = [ "bit-set", "bit-vec", - "bitflags 2.11.1", + "bitflags 2.13.1", "num-traits", - "rand 0.9.4", + "rand 0.9.5", "rand_chacha", "rand_xorshift", "regex-syntax", @@ -2556,9 +2519,9 @@ checksum = "a1d01941d82fa2ab50be1e79e6714289dd7cde78eba4c074bc5a4374f650dfe0" [[package]] name = "quick-xml" -version = "0.40.1" +version = "0.41.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2474bd2e5029e7ccb6abb2ba48cf2383a333851dedf495901544281590c7da7f" +checksum = "e660451e55124f798a69a5af3f49ccfbefbd41910eefd25caf2393e1f3473ec1" dependencies = [ "memchr", "serde", @@ -2566,9 +2529,9 @@ dependencies = [ [[package]] name = "quinn" -version = "0.11.9" +version = "0.11.11" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b9e20a958963c291dc322d98411f541009df2ced7b5a4f2bd52337638cfccf20" +checksum = "0c1a41e437b6bbd489372cd4971de128e85c855f56c57f283d20ff016cf7c0a8" dependencies = [ "bytes", "cfg_aliases", @@ -2578,7 +2541,7 @@ dependencies = [ "rustc-hash", "rustls", "socket2", - "thiserror 2.0.18", + "thiserror 2.0.20", "tokio", "tracing", "web-time", @@ -2586,21 +2549,22 @@ dependencies = [ [[package]] name = "quinn-proto" -version = "0.11.14" +version = "0.11.16" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "434b42fec591c96ef50e21e886936e66d3cc3f737104fdb9b737c40ffb94c098" +checksum = "2f4bfc015262b9df63c8845072ce59068853ff5872180c2ce2f13038b970e560" dependencies = [ "aws-lc-rs", "bytes", - "getrandom 0.3.4", + "getrandom 0.4.3", "lru-slab", - "rand 0.9.4", + "rand 0.10.2", + "rand_pcg", "ring", "rustc-hash", "rustls", "rustls-pki-types", "slab", - "thiserror 2.0.18", + "thiserror 2.0.20", "tinyvec", "tracing", "web-time", @@ -2608,23 +2572,23 @@ dependencies = [ [[package]] name = "quinn-udp" -version = "0.5.14" +version = "0.5.15" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "addec6a0dcad8a8d96a771f815f0eaf55f9d1805756410b39f5fa81332574cbd" +checksum = "35a133f956daabe89a61a685c2649f13d82d5aa4bd5d12d1277e1072a21c0694" dependencies = [ "cfg_aliases", "libc", "once_cell", "socket2", "tracing", - "windows-sys 0.59.0", + "windows-sys 0.61.2", ] [[package]] name = "quote" -version = "1.0.45" +version = "1.0.47" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "41f2619966050689382d2b44f664f4bc593e129785a36d6ee376ddf37259b924" +checksum = "1fbf4db142a473a8d80c26bbf18454ed458bf8d26c8219c331daecfdbd079001" dependencies = [ "proc-macro2", ] @@ -2643,9 +2607,9 @@ checksum = "f8dcc9c7d52a811697d2151c701e0d08956f92b0e24136cf4cf27b57a6a0d9bf" [[package]] name = "rand" -version = "0.9.4" +version = "0.9.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "44c5af06bb1b7d3216d91932aed5265164bf384dc89cd6ba05cf59a35f5f76ea" +checksum = "b9ef1d0d795eb7d84685bca4f72f3649f064e6641543d3a8c415898726a57b41" dependencies = [ "rand_chacha", "rand_core 0.9.5", @@ -2653,12 +2617,12 @@ dependencies = [ [[package]] name = "rand" -version = "0.10.1" +version = "0.10.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d2e8e8bcc7961af1fdac401278c6a831614941f6164ee3bf4ce61b7edb162207" +checksum = "c7f5fa3a058cd35567ef9bfa5e75732bee0f9e4c55fa90477bef2dfcdbc4be80" dependencies = [ "chacha20", - "getrandom 0.4.2", + "getrandom 0.4.3", "rand_core 0.10.1", ] @@ -2687,6 +2651,15 @@ version = "0.10.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "63b8176103e19a2643978565ca18b50549f6101881c443590420e4dc998a3c69" +[[package]] +name = "rand_pcg" +version = "0.10.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "caa0f4137e1c0a72f4c651489402276c8e8e1cf081f3b0ba156d2cbeef09e86a" +dependencies = [ + "rand_core 0.10.1", +] + [[package]] name = "rand_xorshift" version = "0.4.0" @@ -2731,14 +2704,14 @@ version = "0.5.18" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "ed2bf2547551a7053d6fdfafda3f938979645c44812fbfcda098faae3f1a362d" dependencies = [ - "bitflags 2.11.1", + "bitflags 2.13.1", ] [[package]] name = "regex" -version = "1.12.3" +version = "1.13.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e10754a14b9137dd7b1e3e5b0493cc9171fdd105e0ab477f51b72e7f3ac0e276" +checksum = "f020237b6c8eed93db2e2cb53c00c60a8e1bc73da7d073199a1180401450218d" dependencies = [ "aho-corasick", "memchr", @@ -2748,9 +2721,9 @@ dependencies = [ [[package]] name = "regex-automata" -version = "0.4.14" +version = "0.4.18" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6e1dd4122fc1595e8162618945476892eefca7b88c52820e74af6262213cae8f" +checksum = "ad8553b9b26413251cbf30e620595c7a41b3887f03da04579c0e6b0d6a06b4b2" dependencies = [ "aho-corasick", "memchr", @@ -2759,9 +2732,9 @@ dependencies = [ [[package]] name = "regex-syntax" -version = "0.8.10" +version = "0.8.11" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "dc897dd8d9e8bd1ed8cdad82b5966c3e0ecae09fb1907d58efaa013543185d0a" +checksum = "d6f6ff9a378485b298a5286656da665ba74413d36db0979633275d2e708145d4" [[package]] name = "relative-path" @@ -2847,15 +2820,15 @@ dependencies = [ "regex", "relative-path", "rustc_version", - "syn", + "syn 2.0.119", "unicode-ident", ] [[package]] name = "rust_decimal" -version = "1.42.0" +version = "1.42.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0c5108e3d4d903e21aac27f12ba5377b6b34f9f44b325e4894c7924169d06995" +checksum = "be2a24f50780bc85f09cc6ac299bdf1424302742d77221106859c9d8b102126a" dependencies = [ "arrayvec", "num-traits", @@ -2863,15 +2836,15 @@ dependencies = [ [[package]] name = "rustc-demangle" -version = "0.1.27" +version = "0.1.28" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b50b8869d9fc858ce7266cce0194bd74df58b9d0e3f6df3a9fc8eb470d95c09d" +checksum = "b74b56ffa8bb2830709a538c2cbcae9aa062db0d2a42563bfb09bdaae44020eb" [[package]] name = "rustc-hash" -version = "2.1.2" +version = "2.1.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "94300abf3f1ae2e2b8ffb7b58043de3d399c73fa6f4b73826402a5c457614dbe" +checksum = "6b1e7f9a428571be2dc5bc0505c13fb6bf936822b894ec87abf8a08a4e51742d" [[package]] name = "rustc_version" @@ -2888,7 +2861,7 @@ version = "1.1.4" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "b6fe4565b9518b83ef4f91bb47ce29620ca828bd32cb7e408f0062e9930ba190" dependencies = [ - "bitflags 2.11.1", + "bitflags 2.13.1", "errno", "libc", "linux-raw-sys", @@ -2897,9 +2870,9 @@ dependencies = [ [[package]] name = "rustls" -version = "0.23.40" +version = "0.23.43" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ef86cd5876211988985292b91c96a8f2d298df24e75989a43a3c73f2d4d8168b" +checksum = "0283386ce02abc0151e1761d08802dfe86c173b0b494af5cbc086574e453da06" dependencies = [ "aws-lc-rs", "once_cell", @@ -2911,9 +2884,9 @@ dependencies = [ [[package]] name = "rustls-native-certs" -version = "0.8.3" +version = "0.8.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "612460d5f7bea540c490b2b6395d8e34a953e52b491accd6c86c8164c5932a63" +checksum = "dab5152771c58876a2146916e53e35057e1a4dfa2b9df0f0305b07f611fdea4d" dependencies = [ "openssl-probe", "rustls-pki-types", @@ -2923,9 +2896,9 @@ dependencies = [ [[package]] name = "rustls-pki-types" -version = "1.14.1" +version = "1.15.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "30a7197ae7eb376e574fe940d068c30fe0462554a3ddbe4eca7838e049c937a9" +checksum = "2f4925028c7eb5d1fcdaf196971378ed9d2c1c4efc7dc5d011256f76c99c0a96" dependencies = [ "web-time", "zeroize", @@ -2972,9 +2945,9 @@ dependencies = [ [[package]] name = "rustversion" -version = "1.0.22" +version = "1.0.23" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b39cdef0fa800fc44525c84ccb54a029961a8215f9619753635a9c0d2538d46d" +checksum = "cf54715a573b99ac80df0bc206da022bcd442c974952c7b9720069370852e21f" [[package]] name = "rusty-fork" @@ -3035,7 +3008,7 @@ checksum = "1783eabc414609e28a5ba76aee5ddd52199f7107a0b24c2e9746a1ecc34a683d" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] @@ -3044,7 +3017,7 @@ version = "3.7.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "b7f4bc775c73d9a02cde8bf7b2ec4c9d12743edf609006c7facc23998404cd1d" dependencies = [ - "bitflags 2.11.1", + "bitflags 2.13.1", "core-foundation", "core-foundation-sys", "libc", @@ -3073,9 +3046,9 @@ dependencies = [ [[package]] name = "serde" -version = "1.0.228" +version = "1.0.229" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9a8e94ea7f378bd32cbbd37198a4a91436180c5bb472411e48b5ec2e2124ae9e" +checksum = "4148590afebada386688f18773da617792bf2ef03ffc1e4cbd2b1d45b023e0ba" dependencies = [ "serde_core", "serde_derive", @@ -3083,29 +3056,29 @@ dependencies = [ [[package]] name = "serde_core" -version = "1.0.228" +version = "1.0.229" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "41d385c7d4ca58e59fc732af25c3983b67ac852c1a25000afe1175de458b67ad" +checksum = "67dca2c9c51e58a4791a4b1ed58308b39c64224d349a935ab5039aa360942a48" dependencies = [ "serde_derive", ] [[package]] name = "serde_derive" -version = "1.0.228" +version = "1.0.229" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d540f220d3187173da220f885ab66608367b6574e925011a9353e4badda91d79" +checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 3.0.4", ] [[package]] name = "serde_json" -version = "1.0.149" +version = "1.0.151" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "83fc039473c5595ace860d8c4fafa220ff474b3fc6bfdb4293327f1a37e94d86" +checksum = "c841b55ecdae098c80dcae9cf767f6f8a0c2cdb3416bbef72181df4d0fe73f14" dependencies = [ "itoa", "memchr", @@ -3168,9 +3141,9 @@ dependencies = [ [[package]] name = "shlex" -version = "1.3.0" +version = "2.0.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0fda2ff0d084019ba4d7c6f371c95d8fd75ce3524c3cb8fb653a3023f6323e64" +checksum = "f8fadd59c855ef2080decdef8ff161eb6661b86933c9d82e5ba29dc602a55aba" [[package]] name = "signal-hook-registry" @@ -3184,15 +3157,15 @@ dependencies = [ [[package]] name = "simd-adler32" -version = "0.3.9" +version = "0.3.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "703d5c7ef118737c72f1af64ad2f6f8c5e1921f818cdcb97b8fe6fc69bf66214" +checksum = "3a219298ac11a56ea9a6d2120044824d6f01aeb034955e7af7bc16858527deea" [[package]] name = "simd_cesu8" -version = "1.1.1" +version = "1.2.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "94f90157bb87cddf702797c5dadfa0be7d266cdf49e22da2fcaa32eff75b2c33" +checksum = "11031e251abf8611c80f460e19dbdeb54a66db918e49c65a7065b46ac7aec520" dependencies = [ "rustc_version", "simdutf8", @@ -3224,7 +3197,7 @@ checksum = "0c790de23124f9ab44544d7ac05d60440adc586479ce501c1d6d7da3cd8c9cf5" [[package]] name = "slatedb" -version = "0.15.0" +version = "0.16.0" dependencies = [ "ahash", "async-channel", @@ -3233,7 +3206,7 @@ dependencies = [ "backon", "backtrace", "bincode", - "bitflags 2.11.1", + "bitflags 2.13.1", "bytes", "chrono", "crc32fast", @@ -3260,7 +3233,7 @@ dependencies = [ "parking_lot", "pprof", "proptest", - "rand 0.9.4", + "rand 0.9.5", "rstest", "serde", "serde_json", @@ -3286,14 +3259,14 @@ dependencies = [ [[package]] name = "slatedb-bencher" -version = "0.15.0" +version = "0.16.0" dependencies = [ "bytes", "chrono", "clap", "futures", "object_store", - "rand 0.9.4", + "rand 0.9.5", "rand_xorshift", "slatedb", "sysinfo", @@ -3304,7 +3277,7 @@ dependencies = [ [[package]] name = "slatedb-cli" -version = "0.15.0" +version = "0.16.0" dependencies = [ "chrono", "clap", @@ -3324,12 +3297,12 @@ dependencies = [ [[package]] name = "slatedb-common" -version = "0.15.0" +version = "0.16.0" dependencies = [ "chrono", "log", "object_store", - "rand 0.9.4", + "rand 0.9.5", "rand_xoshiro", "serde", "thread_local", @@ -3338,7 +3311,7 @@ dependencies = [ [[package]] name = "slatedb-dst" -version = "0.15.0" +version = "0.16.0" dependencies = [ "async-trait", "bytes", @@ -3349,7 +3322,7 @@ dependencies = [ "log", "object_store", "parking_lot", - "rand 0.9.4", + "rand 0.9.5", "rstest", "slatedb", "slatedb-common", @@ -3363,7 +3336,7 @@ dependencies = [ [[package]] name = "slatedb-txn-obj" -version = "0.15.0" +version = "0.16.0" dependencies = [ "async-trait", "bytes", @@ -3380,7 +3353,7 @@ dependencies = [ [[package]] name = "slatedb-uniffi" -version = "0.15.0" +version = "0.16.0" dependencies = [ "chrono", "figment", @@ -3407,27 +3380,27 @@ checksum = "88414a5ca1f85d82cc34471e975f0f74f6aa54c40f062efa42c0080e7f763f81" [[package]] name = "smallvec" -version = "1.15.1" +version = "1.15.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "67b1b7a3b5fe4f1376887184045fcf45c69e92af734b7aaddc05fb777b6fbd03" +checksum = "8ed6a63f02c8539c91a8685a86f4099661ba3da017932f6ebbea6de3f0fa7c90" [[package]] name = "smawk" -version = "0.3.2" +version = "0.3.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b7c388c1b5e93756d0c740965c41e8822f866621d41acbdf6336a6a168f8840c" +checksum = "e8e2fb0f499abb4d162f2bedad68f5ef91a1682b5a03596ddb67efd37768d100" [[package]] name = "snap" -version = "1.1.1" +version = "1.1.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1b6b67fb9a61334225b5b790716f609cd58395f895b3fe8b328786812a40bc3b" +checksum = "199905e6153d6405f9728fe44daace35f8f837bbf830bb6e85fbd5828709a886" [[package]] name = "socket2" -version = "0.6.3" +version = "0.6.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3a766e1110788c36f4fa1c2b71b387a7815aa65f88ce0229841826633d93723e" +checksum = "c3d1e2c7f27f8d4cb10542a02c49005dbd6e93095799d6f3be745fae9f8fedd4" dependencies = [ "libc", "windows-sys 0.61.2", @@ -3435,9 +3408,9 @@ dependencies = [ [[package]] name = "spin" -version = "0.10.0" +version = "0.10.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d5fe4ccb98d9c292d56fec89a5e07da7fc4cf0dc11e156b41793132775d3e591" +checksum = "023a211cb3138dbc438680b32560ad89f699977624c9f8dbb95a47d5b4c07dd3" dependencies = [ "lock_api", ] @@ -3491,9 +3464,20 @@ dependencies = [ [[package]] name = "syn" -version = "2.0.117" +version = "2.0.119" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e665b8803e7b1d2a727f4023456bbbbe74da67099c585258af0ad9c5013b9b99" +checksum = "872831b642d1a07999a962a351ed35b955ea2cfc8f3862091e2a240a84f17297" +dependencies = [ + "proc-macro2", + "quote", + "unicode-ident", +] + +[[package]] +name = "syn" +version = "3.0.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e6275cddf4610d1775e6d1fe9469b2e77d0f39fd98fb7450901b821e0c53649f" dependencies = [ "proc-macro2", "quote", @@ -3517,7 +3501,7 @@ checksum = "728a70f3dbaf5bab7f0c4b1ac8d7ae5ea60a4b5549c8a5914361c99147a709d2" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] @@ -3547,7 +3531,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "32497e9a4c7b38532efcdebeef879707aa9f794296a4f0244f6f69e9bc8574bd" dependencies = [ "fastrand", - "getrandom 0.4.2", + "getrandom 0.4.3", "once_cell", "rustix", "windows-sys 0.61.2", @@ -3573,11 +3557,11 @@ dependencies = [ [[package]] name = "thiserror" -version = "2.0.18" +version = "2.0.20" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4288b5bcbc7920c07a1149a35cf9590a2aa808e0bc1eafaade0b80947865fbc4" +checksum = "ec86235f5fcc2a73650310756d2ac5b138a5780bbbdfae3eeccec992c435ba4f" dependencies = [ - "thiserror-impl 2.0.18", + "thiserror-impl 2.0.20", ] [[package]] @@ -3588,34 +3572,34 @@ checksum = "4fee6c4efc90059e10f81e6d42c60a18f76588c3d74cb83a0b242a2b6c7504c1" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] name = "thiserror-impl" -version = "2.0.18" +version = "2.0.20" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ebc4ee7f67670e9b64d05fa4253e753e016c6c95ff35b89b7941d6b856dec1d5" +checksum = "bc04cd3e1236dd4a98afca4569f2deb3f120e5422a4023be2cb683f8486292af" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 3.0.4", ] [[package]] name = "thread_local" -version = "1.1.9" +version = "1.1.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f60246a4944f24f6e018aa17cdeffb7818b76356965d03b07d6a9886e8962185" +checksum = "1ad99c4c6d32803332c548b1af0540b357b3f5fc0be8f6c6bfe8b2e6ae784070" dependencies = [ "cfg-if", ] [[package]] name = "time" -version = "0.3.47" +version = "0.3.55" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "743bd48c283afc0388f9b8827b976905fb217ad9e647fae3a379a9283c4def2c" +checksum = "cdb87b95ec50ddfa440816d227a17b2ccbdda963a316a727fda0fc4334f7d134" dependencies = [ "deranged", "num-conv", @@ -3625,9 +3609,9 @@ dependencies = [ [[package]] name = "time-core" -version = "0.1.8" +version = "0.1.9" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7694e1cfe791f8d31026952abf09c69ca6f6fa4e1a1229e18988f06a04a12dca" +checksum = "9e1c906769ad99c88eaa54e728060edef082f8e358ff32030cb7c7d315e81109" [[package]] name = "tinystr" @@ -3651,9 +3635,9 @@ dependencies = [ [[package]] name = "tinyvec" -version = "1.11.0" +version = "1.12.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3e61e67053d25a4e82c844e8424039d9745781b3fc4f32b8d55ed50f5f667ef3" +checksum = "bb4ebadaa0af04fab11ae01eb5f9fdb5f9c5b875506e210e71c07873528baa7f" dependencies = [ "tinyvec_macros", ] @@ -3666,9 +3650,9 @@ checksum = "1f3ccbac311fea05f86f61904b462b55fb3df8837a366dfc601a0161d0532f20" [[package]] name = "tokio" -version = "1.52.3" +version = "1.53.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8fc7f01b389ac15039e4dc9531aa973a135d7a4135281b12d7c1bc79fd57fffe" +checksum = "202caea871b69668250d242070849eb495be178ed697a3e98aebce5bc81a0bed" dependencies = [ "bytes", "libc", @@ -3682,13 +3666,13 @@ dependencies = [ [[package]] name = "tokio-macros" -version = "2.7.0" +version = "2.7.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "385a6cb71ab9ab790c5fe8d67f1645e6c450a7ce006a33de03daa956cf70a496" +checksum = "78773a2a397f451582ce068015985c33193cf6dea8b74d2a639fe457b2f07b0e" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 3.0.4", ] [[package]] @@ -3703,9 +3687,9 @@ dependencies = [ [[package]] name = "tokio-stream" -version = "0.1.18" +version = "0.1.19" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "32da49809aab5c3bc678af03902d4ccddea2a87d028d86392a4b1560c6906c70" +checksum = "a3d06f0b082ba57c26b79407372e57cf2a1e28124f78e9479fe80322cf53420b" dependencies = [ "futures-core", "pin-project-lite", @@ -3725,15 +3709,16 @@ dependencies = [ [[package]] name = "tokio-util" -version = "0.7.18" +version = "0.7.19" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9ae9cec805b01e8fc3fd2fe289f89149a9b66dd16786abd8b19cfa7b48cb0098" +checksum = "494815d09bf52b5548659851081238f0ca39ff638363907596da739561c62c52" dependencies = [ "bytes", "futures-core", "futures-sink", "futures-util", "hashbrown 0.15.5", + "libc", "pin-project-lite", "tokio", ] @@ -3815,16 +3800,16 @@ dependencies = [ "indexmap", "toml_datetime 1.1.1+spec-1.1.0", "toml_parser", - "winnow 1.0.3", + "winnow 1.0.4", ] [[package]] name = "toml_parser" -version = "1.1.2+spec-1.1.0" +version = "1.1.3+spec-1.1.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a2abe9b86193656635d2411dc43050282ca48aa31c2451210f4202550afb7526" +checksum = "1d38ac1cf9b95face32296c0a3ede1fdc270627c9d9c02a7274dd6d960dc4d56" dependencies = [ - "winnow 1.0.3", + "winnow 1.0.4", ] [[package]] @@ -3835,9 +3820,9 @@ checksum = "5d99f8c9a7727884afe522e9bd5edbfc91a3312b36a77b5fb8926e4c31a41801" [[package]] name = "toml_writer" -version = "1.1.1+spec-1.1.0" +version = "1.1.2+spec-1.1.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "756daf9b1013ebe47a8776667b466417e2d4c5679d441c26230efd9ef78692db" +checksum = "7d56353a2a665ad0f41a421187180aab746c8c325620617ad883a99a1cbe66d2" [[package]] name = "tower" @@ -3860,7 +3845,7 @@ version = "0.6.11" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "4cfcf7e2740e6fc6d4d688b4ef00650406bb94adf4731e43c096c3a19fe40840" dependencies = [ - "bitflags 2.11.1", + "bitflags 2.13.1", "bytes", "futures-util", "http", @@ -3904,7 +3889,7 @@ checksum = "7490cfa5ec963746568740651ac6781f701c9c5ea257c58e057f3ba8cf69e8da" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] @@ -3954,18 +3939,18 @@ checksum = "e421abadd41a4225275504ea4d6566923418b7f05506fbc9c0fe86ba7396114b" [[package]] name = "twox-hash" -version = "2.1.2" +version = "2.1.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9ea3136b675547379c4bd395ca6b938e5ad3c3d20fad76e7fe85f9e0d011419c" +checksum = "8464ec13c3691491391d9fce00f6416c9a48e46972f72d7865688be2080192c9" dependencies = [ - "rand 0.9.4", + "rand 0.10.2", ] [[package]] name = "typenum" -version = "1.20.0" +version = "1.20.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "40ce102ab67701b8526c123c1bab5cbe42d7040ccfd0f64af1a385808d2f43de" +checksum = "b6f5e870be6c3b371b77fe0ee0bafb859fa4964b4404c27de1d380043c4dda20" [[package]] name = "ulid" @@ -3973,7 +3958,7 @@ version = "1.2.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "470dbf6591da1b39d43c14523b2b469c86879a53e8b758c8e090a470fe7b1fbe" dependencies = [ - "rand 0.9.4", + "rand 0.9.5", "serde", "web-time", ] @@ -3999,17 +3984,11 @@ version = "1.0.24" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "e6e4313cd5fcd3dad5cafa179702e2b244f760991f45397d14d4ebf38247da75" -[[package]] -name = "unicode-xid" -version = "0.2.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ebc1c04c71510c7f702b52b7c350734c9ff1295c464a03335b00bb84fc54f853" - [[package]] name = "uniffi" -version = "0.31.1" +version = "0.31.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "dc5f2297ee5b893405bed1a6929faec4713a061df158ecf5198089f23910d470" +checksum = "46eefd5468602930da46b1f49d3448c6dfc2e81295f93120f23f8174fd70267f" dependencies = [ "anyhow", "cargo_metadata", @@ -4021,9 +4000,9 @@ dependencies = [ [[package]] name = "uniffi_bindgen" -version = "0.31.1" +version = "0.31.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8bc0c60a9607e7ab77a2ad47ec5530178015014839db25af7512447d2238016c" +checksum = "c4a0c9b375d32e1365cdb2bdd7cb495eecf6fac851ddbad077412b4ee1888514" dependencies = [ "anyhow", "askama", @@ -4047,9 +4026,9 @@ dependencies = [ [[package]] name = "uniffi_core" -version = "0.31.1" +version = "0.31.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "77baf5d539fe2e1ad6805e942dbc5dbdeb2b83eb5f2b3a6535d422ca4b02a12f" +checksum = "eec017b112701681f6fbbe5d92014b5c468eb0b177a94389de03ceec40665095" dependencies = [ "anyhow", "async-compat", @@ -4060,22 +4039,22 @@ dependencies = [ [[package]] name = "uniffi_internal_macros" -version = "0.31.1" +version = "0.31.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b4b42137524f4be6400fcaca9d02c1d4ecb6ad917e4013c0b93235526d8396e5" +checksum = "4641669b48fefbc5e80ff08c5004d9c7617fb91232131a6734ab6712779cb04c" dependencies = [ "anyhow", "indexmap", "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] name = "uniffi_macros" -version = "0.31.1" +version = "0.31.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d9273ec45330d8fe9a3701b7b983cea7a4e218503359831967cb95d26b873561" +checksum = "eeb8617ee814de22caf7417bf514715ba0b3f46bd9d5a5d794413fd8282cb737" dependencies = [ "camino", "fs-err", @@ -4083,16 +4062,16 @@ dependencies = [ "proc-macro2", "quote", "serde", - "syn", + "syn 2.0.119", "toml 0.9.12+spec-1.1.0", "uniffi_meta", ] [[package]] name = "uniffi_meta" -version = "0.31.1" +version = "0.31.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "431d2f443e7828a6c29d188de98b6771a6491ee98bba2d4372643bf93f988a18" +checksum = "58d5b94fc92803d21b2928bd15c6f06e57609b95caf98ea561c99cda1b6d2a25" dependencies = [ "anyhow", "siphasher", @@ -4102,9 +4081,9 @@ dependencies = [ [[package]] name = "uniffi_pipeline" -version = "0.31.1" +version = "0.31.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "761ef74f6175e15603d0424cc5f98854c5baccfe7bf4ccb08e5816f9ab8af689" +checksum = "032739b3ec725576914c15899dedaf080163ced86b6934566c20ec2b20ce90ca" dependencies = [ "anyhow", "heck 0.5.0", @@ -4115,9 +4094,9 @@ dependencies = [ [[package]] name = "uniffi_udl" -version = "0.31.1" +version = "0.31.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "68773ec0e1c067b6505a73bbf6a5782f31a7f9209333a0df97b87565c46bf370" +checksum = "fc0a1d0a0252ce1af9e8ce78ba67ac0d8937fb2bedaf10cbddd43d3614d06ec6" dependencies = [ "anyhow", "textwrap", @@ -4163,11 +4142,11 @@ checksum = "06abde3611657adf66d383f00b093d7faecc7fa57071cce2578660c9f1010821" [[package]] name = "uuid" -version = "1.23.1" +version = "1.24.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ddd74a9687298c6858e9b88ec8935ec45d22e8fd5e6394fa1bd4e99a87789c76" +checksum = "bf3923a6f5c4c6382e0b653c4117f48d631ea17f38ed86e2a828e6f7412f5239" dependencies = [ - "getrandom 0.4.2", + "getrandom 0.4.3", "js-sys", "serde_core", "wasm-bindgen", @@ -4221,27 +4200,18 @@ checksum = "ccf3ec651a847eb01de73ccad15eb7d99f80485de043efb2f370cd654f4ea44b" [[package]] name = "wasip2" -version = "1.0.3+wasi-0.2.9" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "20064672db26d7cdc89c7798c48a0fdfac8213434a1186e5ef29fd560ae223d6" -dependencies = [ - "wit-bindgen 0.57.1", -] - -[[package]] -name = "wasip3" -version = "0.4.0+wasi-0.3.0-rc-2026-01-06" +version = "1.0.4+wasi-0.2.12" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5428f8bf88ea5ddc08faddef2ac4a67e390b88186c703ce6dbd955e1c145aca5" +checksum = "b67efb37e106e55ce722a510d6b5f9c17f083e5fc79afc2badeb12cc313d9487" dependencies = [ - "wit-bindgen 0.51.0", + "wit-bindgen", ] [[package]] name = "wasm-bindgen" -version = "0.2.121" +version = "0.2.126" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "49ace1d07c165b0864824eee619580c4689389afa9dc9ed3a4c75040d82e6790" +checksum = "4b067c0c11094aef6b7a801c1e34a26affafdf3d051dba08456b868789aaf9a4" dependencies = [ "cfg-if", "once_cell", @@ -4252,9 +4222,9 @@ dependencies = [ [[package]] name = "wasm-bindgen-futures" -version = "0.4.71" +version = "0.4.76" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "96492d0d3ffba25305a7dc88720d250b1401d7edca02cc3bcd50633b424673b8" +checksum = "c62df1340f32221cb9c54d6a27b030e3dba64361d4a95bed55f9aacb44da291d" dependencies = [ "js-sys", "wasm-bindgen", @@ -4262,9 +4232,9 @@ dependencies = [ [[package]] name = "wasm-bindgen-macro" -version = "0.2.121" +version = "0.2.126" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8e68e6f4afd367a562002c05637acb8578ff2dea1943df76afb9e83d177c8578" +checksum = "167ce5e579f6bcf889c4f7175a8a5a585de84e8ff93976ce393efa5f2837aab1" dependencies = [ "quote", "wasm-bindgen-macro-support", @@ -4272,48 +4242,26 @@ dependencies = [ [[package]] name = "wasm-bindgen-macro-support" -version = "0.2.121" +version = "0.2.126" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d95a9ec35c64b2a7cb35d3fead40c4238d0940c86d107136999567a4703259f2" +checksum = "f3997c7839262f4ef12cf90b818d6340c18e80f263f1a94bf157d0ec4420380e" dependencies = [ "bumpalo", "proc-macro2", "quote", - "syn", + "syn 2.0.119", "wasm-bindgen-shared", ] [[package]] name = "wasm-bindgen-shared" -version = "0.2.121" +version = "0.2.126" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c4e0100b01e9f0d03189a92b96772a1fb998639d981193d7dbab487302513441" +checksum = "dc1b4cb0cc549fcf58d7dfc081778139b3d283a081644e833e84682ad71cea24" dependencies = [ "unicode-ident", ] -[[package]] -name = "wasm-encoder" -version = "0.244.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "990065f2fe63003fe337b932cfb5e3b80e0b4d0f5ff650e6985b1048f62c8319" -dependencies = [ - "leb128fmt", - "wasmparser", -] - -[[package]] -name = "wasm-metadata" -version = "0.244.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bb0e353e6a2fbdc176932bbaab493762eb1255a7900fe0fea1a2f96c296cc909" -dependencies = [ - "anyhow", - "indexmap", - "wasm-encoder", - "wasmparser", -] - [[package]] name = "wasm-streams" version = "0.5.0" @@ -4327,23 +4275,11 @@ dependencies = [ "web-sys", ] -[[package]] -name = "wasmparser" -version = "0.244.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "47b807c72e1bac69382b3a6fb3dbe8ea4c0ed87ff5629b8685ae6b9a611028fe" -dependencies = [ - "bitflags 2.11.1", - "hashbrown 0.15.5", - "indexmap", - "semver", -] - [[package]] name = "web-sys" -version = "0.3.98" +version = "0.3.103" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4b572dff8bcf38bad0fa19729c89bb5748b2b9b1d8be70cf90df697e3a8f32aa" +checksum = "8622dcb61c0bcc9fffa6938bed81210af2da9a7e4a1a834b2e37a59b6dfb6141" dependencies = [ "js-sys", "wasm-bindgen", @@ -4361,9 +4297,9 @@ dependencies = [ [[package]] name = "webpki-root-certs" -version = "1.0.8" +version = "1.0.9" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0d46a5a140e6f7afeccd8eae97eff335163939eac8b929834875168b29b3d267" +checksum = "b96554aa2acc8ccdb7e1c9a58a7a68dd5d13bccc69cd124cb09406db612a1c9b" dependencies = [ "rustls-pki-types", ] @@ -4415,7 +4351,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "9babd3a767a4c1aef6900409f85f5d53ce2544ccdfaa86dad48c91782c6d6893" dependencies = [ "windows-collections", - "windows-core", + "windows-core 0.61.2", "windows-future", "windows-link 0.1.3", "windows-numerics", @@ -4427,7 +4363,7 @@ version = "0.2.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "3beeceb5e5cfd9eb1d76b381630e82c4241ccd0d27f1a39ed41b2760b255c5e8" dependencies = [ - "windows-core", + "windows-core 0.61.2", ] [[package]] @@ -4439,8 +4375,21 @@ dependencies = [ "windows-implement", "windows-interface", "windows-link 0.1.3", - "windows-result", - "windows-strings", + "windows-result 0.3.4", + "windows-strings 0.4.2", +] + +[[package]] +name = "windows-core" +version = "0.62.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b8e83a14d34d0623b51dce9581199302a221863196a1dde71a7663a4c2be9deb" +dependencies = [ + "windows-implement", + "windows-interface", + "windows-link 0.2.1", + "windows-result 0.4.1", + "windows-strings 0.5.1", ] [[package]] @@ -4449,7 +4398,7 @@ version = "0.2.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "fc6a41e98427b19fe4b73c550f060b59fa592d7d686537eebf9385621bfbad8e" dependencies = [ - "windows-core", + "windows-core 0.61.2", "windows-link 0.1.3", "windows-threading", ] @@ -4462,7 +4411,7 @@ checksum = "053e2e040ab57b9dc951b72c264860db7eb3b0200ba345b4e4c3b14f67855ddf" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] @@ -4473,7 +4422,7 @@ checksum = "3f316c4a2570ba26bbec722032c4099d8c8bc095efccdc15688708623367e358" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] @@ -4494,7 +4443,7 @@ version = "0.2.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "9150af68066c4c5c07ddc0ce30421554771e528bde427614c61038bc2c92c2b1" dependencies = [ - "windows-core", + "windows-core 0.61.2", "windows-link 0.1.3", ] @@ -4507,6 +4456,15 @@ dependencies = [ "windows-link 0.1.3", ] +[[package]] +name = "windows-result" +version = "0.4.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7781fa89eaf60850ac3d2da7af8e5242a5ea78d1a11c49bf2910bb5a73853eb5" +dependencies = [ + "windows-link 0.2.1", +] + [[package]] name = "windows-strings" version = "0.4.2" @@ -4516,6 +4474,15 @@ dependencies = [ "windows-link 0.1.3", ] +[[package]] +name = "windows-strings" +version = "0.5.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7837d08f69c77cf6b07689544538e017c1bfcf57e34b4c0ff58e6c2cd3b37091" +dependencies = [ + "windows-link 0.2.1", +] + [[package]] name = "windows-sys" version = "0.52.0" @@ -4636,107 +4603,19 @@ dependencies = [ [[package]] name = "winnow" -version = "1.0.3" +version = "1.0.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0592e1c9d151f854e6fd382574c3a0855250e1d9b2f99d9281c6e6391af352f1" +checksum = "23b97319f7b8343df12cc98938e5c3eb436064524c8d2b4e30a1d3a36eecdf81" dependencies = [ "memchr", ] -[[package]] -name = "wit-bindgen" -version = "0.51.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d7249219f66ced02969388cf2bb044a09756a083d0fab1e566056b04d9fbcaa5" -dependencies = [ - "wit-bindgen-rust-macro", -] - [[package]] name = "wit-bindgen" version = "0.57.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "1ebf944e87a7c253233ad6766e082e3cd714b5d03812acc24c318f549614536e" -[[package]] -name = "wit-bindgen-core" -version = "0.51.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ea61de684c3ea68cb082b7a88508a8b27fcc8b797d738bfc99a82facf1d752dc" -dependencies = [ - "anyhow", - "heck 0.5.0", - "wit-parser", -] - -[[package]] -name = "wit-bindgen-rust" -version = "0.51.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b7c566e0f4b284dd6561c786d9cb0142da491f46a9fbed79ea69cdad5db17f21" -dependencies = [ - "anyhow", - "heck 0.5.0", - "indexmap", - "prettyplease", - "syn", - "wasm-metadata", - "wit-bindgen-core", - "wit-component", -] - -[[package]] -name = "wit-bindgen-rust-macro" -version = "0.51.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0c0f9bfd77e6a48eccf51359e3ae77140a7f50b1e2ebfe62422d8afdaffab17a" -dependencies = [ - "anyhow", - "prettyplease", - "proc-macro2", - "quote", - "syn", - "wit-bindgen-core", - "wit-bindgen-rust", -] - -[[package]] -name = "wit-component" -version = "0.244.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9d66ea20e9553b30172b5e831994e35fbde2d165325bec84fc43dbf6f4eb9cb2" -dependencies = [ - "anyhow", - "bitflags 2.11.1", - "indexmap", - "log", - "serde", - "serde_derive", - "serde_json", - "wasm-encoder", - "wasm-metadata", - "wasmparser", - "wit-parser", -] - -[[package]] -name = "wit-parser" -version = "0.244.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ecc8ac4bc1dc3381b7f59c34f00b67e18f910c2c0f50015669dde7def656a736" -dependencies = [ - "anyhow", - "id-arena", - "indexmap", - "log", - "semver", - "serde", - "serde_derive", - "serde_json", - "unicode-xid", - "wasmparser", -] - [[package]] name = "writeable" version = "0.6.3" @@ -4751,9 +4630,9 @@ checksum = "cfe53a6657fd280eaa890a3bc59152892ffa3e30101319d168b781ed6529b049" [[package]] name = "yoke" -version = "0.8.2" +version = "0.8.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "abe8c5fda708d9ca3df187cae8bfb9ceda00dd96231bed36e445a1a48e66f9ca" +checksum = "709fe23a0424b6a435d82152b1bd3fdfb0833487d5fa90d05d42762a9891fef5" dependencies = [ "stable_deref_trait", "yoke-derive", @@ -4768,28 +4647,28 @@ checksum = "de844c262c8848816172cef550288e7dc6c7b7814b4ee56b3e1553f275f1858e" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.119", "synstructure", ] [[package]] name = "zerocopy" -version = "0.8.48" +version = "0.8.56" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "eed437bf9d6692032087e337407a86f04cd8d6a16a37199ed57949d415bd68e9" +checksum = "556764e583adb45a9f8d413c2a147fa7e8d821e48e12b14fd560b607998b75eb" dependencies = [ "zerocopy-derive", ] [[package]] name = "zerocopy-derive" -version = "0.8.48" +version = "0.8.56" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "70e3cd084b1788766f53af483dd21f93881ff30d7320490ec3ef7526d203bad4" +checksum = "f2ab42fc20575779bd240faa45f94a74256f755c0fa9e89f0ede20d91d0cdfc1" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] @@ -4809,15 +4688,15 @@ checksum = "11532158c46691caf0f2593ea8358fed6bbf68a0315e80aae9bd41fbade684a1" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.119", "synstructure", ] [[package]] name = "zeroize" -version = "1.8.2" +version = "1.9.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b97154e67e32c85465826e8bcc1c59429aaaf107c1e4a9e53c8d8ccd5eff88d0" +checksum = "e13c156562582aa81c60cb29407084cdb54c4164760106ab78e6c5b0858cf64e" [[package]] name = "zerotrie" @@ -4849,14 +4728,14 @@ checksum = "625dc425cab0dca6dc3c3319506e6593dcb08a9f387ea3b284dbd52a92c40555" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] name = "zmij" -version = "1.0.21" +version = "1.0.23" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b8848ee67ecc8aedbaf3e4122217aff892639231befc6a1b58d29fff4c2cabaa" +checksum = "29666d0abbfad1e3dc4dcf6144730dd3a3ab225bbbdac83319345b1b44ccfc1b" [[package]] name = "zstd" diff --git a/Cargo.toml b/Cargo.toml index f601ff5855..729335c05a 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -12,7 +12,7 @@ members = [ ] [workspace.package] -version = "0.15.0" +version = "0.16.0" edition = "2021" repository = "https://github.com/slatedb/slatedb" license = "Apache-2.0" @@ -71,9 +71,9 @@ serde = "1.0" serde_json = "1.0.142" siphasher = "1" smallvec = "1.15.1" -slatedb = { path = "slatedb", version = "0.15.0" } -slatedb-common = { path = "slatedb-common", version = "0.15.0" } -slatedb-txn-obj = { path = "slatedb-txn-obj", version = "0.15.0" } +slatedb = { path = "slatedb", version = "0.16.0" } +slatedb-common = { path = "slatedb-common", version = "0.16.0" } +slatedb-txn-obj = { path = "slatedb-txn-obj", version = "0.16.0" } snap = "1.1.1" sysinfo = "0.35.2" thiserror = "1.0.63" @@ -112,3 +112,4 @@ unexpected_cfgs = { level = "allow", check-cfg = [ 'cfg(tokio_unstable)', ] } unreachable_pub = "warn" +unused_qualifications = "warn" diff --git a/README.md b/README.md index a15dc595d2..54ce988b10 100644 --- a/README.md +++ b/README.md @@ -16,7 +16,7 @@ To mitigate high write API costs (PUTs), SlateDB batches writes. Rather than writing every `put()` call to object storage, MemTables are flushed periodically to object storage as a string-sorted table (SST). The flush interval is configurable. -`put()` returns a `Future` that resolves when the data is durably persisted. Clients that prefer lower latency at the cost of durability can instead use `put_with_options` with `await_durable` set to `false`. +Write operations return a `WriteHandle` after updating the in-memory WAL and MemTable. Call `handle.await_durable().await` to wait for one write to become durable, or call `db.flush().await` to flush all pending writes. To mitigate read latency and read API costs (GETs), SlateDB will use standard LSM-tree caching techniques: in-memory block caches, compression, bloom filters, and local SST disk caches. @@ -152,6 +152,7 @@ SlateDB follows Semantic Versioning. We release new versions approximately every See who's using SlateDB. +- [4og.io](https://4og.io) - [Dropbox](https://www.dropbox.com) - [Embucket](https://www.embucket.com) - [Gadget](https://gadget.dev) @@ -161,15 +162,16 @@ See who's using SlateDB. - [Massive](https://massive.com) - [Merklemap](https://merklemap.com) - [OpenData](https://www.opendata.dev) +- [Prisma](https://www.prisma.io) - [Responsive](https://responsive.dev) - [s2-lite](https://github.com/s2-streamstore/s2) -- [SQLync](https://sqlync.com) - [Storrito](https://storrito.com) - [Taquba](https://github.com/micllam/taquba) - [Tensorlake](https://www.tensorlake.ai) - [Volga](https://github.com/volga-project/volga) - [WombatKV](https://github.com/Venkat2811/wombatkv) - [ZeroFS](https://zerofs.net) +- [LixRay](https://lixray.com) ## Talks diff --git a/bindings/go/uniffi/doc.go b/bindings/go/uniffi/doc.go index a3a5daad72..cd8744dfaf 100644 --- a/bindings/go/uniffi/doc.go +++ b/bindings/go/uniffi/doc.go @@ -100,14 +100,14 @@ // [Db.Delete], and [Db.Merge], plus batch and durability controls through // [PutOptions], [MergeOptions], [WriteOptions], and [FlushOptions]. // -// [WriteHandle] reports metadata assigned to a successful write, including the -// sequence number and creation timestamp. +// [WriteHandle] reports metadata assigned to a successful write and exposes +// [WriteHandle.AwaitDurable] for waiting until that specific write is durable. // // [WriteBatch] collects multiple mutations and applies them atomically through // [Db.Write] or [Db.WriteWithOptions]. Batches are single-use once submitted. // // TTL behavior is configured with [Ttl] implementations such as [TtlDefault], -// [TtlNoExpiry], and [TtlExpireAfterTicks]. +// [TtlNoExpiry], and [TtlExpireAfterMillis]. // // # Transactions // @@ -135,13 +135,15 @@ // interface. Rust-side logging can be forwarded into Go code with // [InitLogging] and a [LogCallback]. // -// # WAL Inspection +// # Change Data Capture // -// [NewWalReader] opens a [WalReader] for inspecting WAL files under a database -// path. [WalReader.List] enumerates [WalFile] handles, [WalFile.Metadata] -// returns object-store metadata, and [WalFile.Iterator] returns a -// [WalFileIterator] that yields raw [RowEntry] values. This is primarily useful -// for debugging, diagnostics, and low-level tooling. +// [NewSlateDbWalReader] opens a [SlateDbWalReader] for live WAL streaming. +// Call [SlateDbWalReader.Iterator] once with the first unconsumed WAL file ID, +// then keep calling [SlateDbWalIterator.Next]. The iterator waits and polls +// internally at the current tail. Persist every [WalRows.LastConsumedWalFileId], +// including empty fence batches, and resume from the following ID after a +// restart. [SlateDbWalReader.LastWalFileId] is available when a snapshot of the +// current tail is useful, but is not needed to drive the stream. // // # Errors // @@ -159,8 +161,8 @@ // // Most exported handle types own a Rust-side resource and provide an explicit // `Destroy` method, including [ObjectStore], [DbBuilder], [Db], [DbReader], -// [DbSnapshot], [DbTransaction], [DbIterator], [WalReader], [WalFile], -// [WalFileIterator], [Settings], and [WriteBatch]. +// [DbSnapshot], [DbTransaction], [DbIterator], [SlateDbWalReader], +// [SlateDbWalIterator], [Settings], and [WriteBatch]. // // These handles install Go finalizers, but callers should not rely on garbage // collection for timely cleanup. Prefer calling `Destroy` explicitly when a @@ -169,6 +171,7 @@ // binding handle. // // Builders are single-use after `Build`. [WriteBatch] is single-use after -// `Write`. Iterator `Next` methods return `nil` when exhausted, and transaction -// commit methods may return `nil` when no write was emitted. +// `Write`. Bounded iterator `Next` methods return `nil` when exhausted; the live +// [SlateDbWalIterator] instead waits at the current tail. Transaction commit +// methods may return `nil` when no write was emitted. package slatedb diff --git a/bindings/go/uniffi/slatedb.go b/bindings/go/uniffi/slatedb.go index afa3390e8d..af4e20de97 100644 --- a/bindings/go/uniffi/slatedb.go +++ b/bindings/go/uniffi/slatedb.go @@ -799,7 +799,7 @@ func uniffiCheckChecksums() { checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { return C.uniffi_slatedb_uniffi_checksum_method_db_delete() }) - if checksum != 4063 { + if checksum != 29763 { // If this happens try cleaning and rebuilding your project panic("slatedb: uniffi_slatedb_uniffi_checksum_method_db_delete: UniFFI API checksum mismatch") } @@ -808,7 +808,7 @@ func uniffiCheckChecksums() { checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { return C.uniffi_slatedb_uniffi_checksum_method_db_delete_with_options() }) - if checksum != 44744 { + if checksum != 47162 { // If this happens try cleaning and rebuilding your project panic("slatedb: uniffi_slatedb_uniffi_checksum_method_db_delete_with_options: UniFFI API checksum mismatch") } @@ -880,7 +880,7 @@ func uniffiCheckChecksums() { checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { return C.uniffi_slatedb_uniffi_checksum_method_db_merge() }) - if checksum != 28366 { + if checksum != 37097 { // If this happens try cleaning and rebuilding your project panic("slatedb: uniffi_slatedb_uniffi_checksum_method_db_merge: UniFFI API checksum mismatch") } @@ -889,7 +889,7 @@ func uniffiCheckChecksums() { checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { return C.uniffi_slatedb_uniffi_checksum_method_db_merge_with_options() }) - if checksum != 15865 { + if checksum != 37495 { // If this happens try cleaning and rebuilding your project panic("slatedb: uniffi_slatedb_uniffi_checksum_method_db_merge_with_options: UniFFI API checksum mismatch") } @@ -898,7 +898,7 @@ func uniffiCheckChecksums() { checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { return C.uniffi_slatedb_uniffi_checksum_method_db_put() }) - if checksum != 53275 { + if checksum != 2894 { // If this happens try cleaning and rebuilding your project panic("slatedb: uniffi_slatedb_uniffi_checksum_method_db_put: UniFFI API checksum mismatch") } @@ -907,7 +907,7 @@ func uniffiCheckChecksums() { checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { return C.uniffi_slatedb_uniffi_checksum_method_db_put_with_options() }) - if checksum != 37591 { + if checksum != 12036 { // If this happens try cleaning and rebuilding your project panic("slatedb: uniffi_slatedb_uniffi_checksum_method_db_put_with_options: UniFFI API checksum mismatch") } @@ -957,6 +957,15 @@ func uniffiCheckChecksums() { panic("slatedb: uniffi_slatedb_uniffi_checksum_method_db_shutdown: UniFFI API checksum mismatch") } } + { + checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { + return C.uniffi_slatedb_uniffi_checksum_method_db_shutdown_with_options() + }) + if checksum != 54951 { + // If this happens try cleaning and rebuilding your project + panic("slatedb: uniffi_slatedb_uniffi_checksum_method_db_shutdown_with_options: UniFFI API checksum mismatch") + } + } { checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { return C.uniffi_slatedb_uniffi_checksum_method_db_snapshot() @@ -988,7 +997,7 @@ func uniffiCheckChecksums() { checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { return C.uniffi_slatedb_uniffi_checksum_method_db_write() }) - if checksum != 29016 { + if checksum != 63711 { // If this happens try cleaning and rebuilding your project panic("slatedb: uniffi_slatedb_uniffi_checksum_method_db_write: UniFFI API checksum mismatch") } @@ -997,7 +1006,7 @@ func uniffiCheckChecksums() { checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { return C.uniffi_slatedb_uniffi_checksum_method_db_write_with_options() }) - if checksum != 13580 { + if checksum != 36986 { // If this happens try cleaning and rebuilding your project panic("slatedb: uniffi_slatedb_uniffi_checksum_method_db_write_with_options: UniFFI API checksum mismatch") } @@ -1186,7 +1195,7 @@ func uniffiCheckChecksums() { checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { return C.uniffi_slatedb_uniffi_checksum_method_dbtransaction_commit() }) - if checksum != 56467 { + if checksum != 56426 { // If this happens try cleaning and rebuilding your project panic("slatedb: uniffi_slatedb_uniffi_checksum_method_dbtransaction_commit: UniFFI API checksum mismatch") } @@ -1195,7 +1204,7 @@ func uniffiCheckChecksums() { checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { return C.uniffi_slatedb_uniffi_checksum_method_dbtransaction_commit_with_options() }) - if checksum != 62589 { + if checksum != 5743 { // If this happens try cleaning and rebuilding your project panic("slatedb: uniffi_slatedb_uniffi_checksum_method_dbtransaction_commit_with_options: UniFFI API checksum mismatch") } @@ -1573,7 +1582,7 @@ func uniffiCheckChecksums() { checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { return C.uniffi_slatedb_uniffi_checksum_method_settings_set() }) - if checksum != 34344 { + if checksum != 16989 { // If this happens try cleaning and rebuilding your project panic("slatedb: uniffi_slatedb_uniffi_checksum_method_settings_set: UniFFI API checksum mismatch") } @@ -1589,119 +1598,101 @@ func uniffiCheckChecksums() { } { checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { - return C.uniffi_slatedb_uniffi_checksum_method_walfile_id() - }) - if checksum != 62512 { - // If this happens try cleaning and rebuilding your project - panic("slatedb: uniffi_slatedb_uniffi_checksum_method_walfile_id: UniFFI API checksum mismatch") - } - } - { - checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { - return C.uniffi_slatedb_uniffi_checksum_method_walfile_iterator() - }) - if checksum != 46880 { - // If this happens try cleaning and rebuilding your project - panic("slatedb: uniffi_slatedb_uniffi_checksum_method_walfile_iterator: UniFFI API checksum mismatch") - } - } - { - checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { - return C.uniffi_slatedb_uniffi_checksum_method_walfile_metadata() + return C.uniffi_slatedb_uniffi_checksum_method_slatedbwaliterator_next() }) - if checksum != 45103 { + if checksum != 46461 { // If this happens try cleaning and rebuilding your project - panic("slatedb: uniffi_slatedb_uniffi_checksum_method_walfile_metadata: UniFFI API checksum mismatch") + panic("slatedb: uniffi_slatedb_uniffi_checksum_method_slatedbwaliterator_next: UniFFI API checksum mismatch") } } { checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { - return C.uniffi_slatedb_uniffi_checksum_method_walfile_next_file() + return C.uniffi_slatedb_uniffi_checksum_method_slatedbwalreader_iterator() }) - if checksum != 56800 { + if checksum != 10327 { // If this happens try cleaning and rebuilding your project - panic("slatedb: uniffi_slatedb_uniffi_checksum_method_walfile_next_file: UniFFI API checksum mismatch") + panic("slatedb: uniffi_slatedb_uniffi_checksum_method_slatedbwalreader_iterator: UniFFI API checksum mismatch") } } { checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { - return C.uniffi_slatedb_uniffi_checksum_method_walfile_next_id() + return C.uniffi_slatedb_uniffi_checksum_method_slatedbwalreader_last_wal_file_id() }) - if checksum != 48353 { + if checksum != 40190 { // If this happens try cleaning and rebuilding your project - panic("slatedb: uniffi_slatedb_uniffi_checksum_method_walfile_next_id: UniFFI API checksum mismatch") + panic("slatedb: uniffi_slatedb_uniffi_checksum_method_slatedbwalreader_last_wal_file_id: UniFFI API checksum mismatch") } } { checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { - return C.uniffi_slatedb_uniffi_checksum_method_walfileiterator_next() + return C.uniffi_slatedb_uniffi_checksum_method_writebatch_delete() }) - if checksum != 51490 { + if checksum != 58549 { // If this happens try cleaning and rebuilding your project - panic("slatedb: uniffi_slatedb_uniffi_checksum_method_walfileiterator_next: UniFFI API checksum mismatch") + panic("slatedb: uniffi_slatedb_uniffi_checksum_method_writebatch_delete: UniFFI API checksum mismatch") } } { checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { - return C.uniffi_slatedb_uniffi_checksum_method_walreader_get() + return C.uniffi_slatedb_uniffi_checksum_method_writebatch_merge() }) - if checksum != 11510 { + if checksum != 62067 { // If this happens try cleaning and rebuilding your project - panic("slatedb: uniffi_slatedb_uniffi_checksum_method_walreader_get: UniFFI API checksum mismatch") + panic("slatedb: uniffi_slatedb_uniffi_checksum_method_writebatch_merge: UniFFI API checksum mismatch") } } { checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { - return C.uniffi_slatedb_uniffi_checksum_method_walreader_list() + return C.uniffi_slatedb_uniffi_checksum_method_writebatch_merge_with_options() }) - if checksum != 43661 { + if checksum != 24696 { // If this happens try cleaning and rebuilding your project - panic("slatedb: uniffi_slatedb_uniffi_checksum_method_walreader_list: UniFFI API checksum mismatch") + panic("slatedb: uniffi_slatedb_uniffi_checksum_method_writebatch_merge_with_options: UniFFI API checksum mismatch") } } { checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { - return C.uniffi_slatedb_uniffi_checksum_method_writebatch_delete() + return C.uniffi_slatedb_uniffi_checksum_method_writebatch_put() }) - if checksum != 58549 { + if checksum != 48246 { // If this happens try cleaning and rebuilding your project - panic("slatedb: uniffi_slatedb_uniffi_checksum_method_writebatch_delete: UniFFI API checksum mismatch") + panic("slatedb: uniffi_slatedb_uniffi_checksum_method_writebatch_put: UniFFI API checksum mismatch") } } { checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { - return C.uniffi_slatedb_uniffi_checksum_method_writebatch_merge() + return C.uniffi_slatedb_uniffi_checksum_method_writebatch_put_with_options() }) - if checksum != 62067 { + if checksum != 31177 { // If this happens try cleaning and rebuilding your project - panic("slatedb: uniffi_slatedb_uniffi_checksum_method_writebatch_merge: UniFFI API checksum mismatch") + panic("slatedb: uniffi_slatedb_uniffi_checksum_method_writebatch_put_with_options: UniFFI API checksum mismatch") } } { checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { - return C.uniffi_slatedb_uniffi_checksum_method_writebatch_merge_with_options() + return C.uniffi_slatedb_uniffi_checksum_method_writehandle_await_durable() }) - if checksum != 24696 { + if checksum != 2953 { // If this happens try cleaning and rebuilding your project - panic("slatedb: uniffi_slatedb_uniffi_checksum_method_writebatch_merge_with_options: UniFFI API checksum mismatch") + panic("slatedb: uniffi_slatedb_uniffi_checksum_method_writehandle_await_durable: UniFFI API checksum mismatch") } } { checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { - return C.uniffi_slatedb_uniffi_checksum_method_writebatch_put() + return C.uniffi_slatedb_uniffi_checksum_method_writehandle_create_ts() }) - if checksum != 48246 { + if checksum != 16841 { // If this happens try cleaning and rebuilding your project - panic("slatedb: uniffi_slatedb_uniffi_checksum_method_writebatch_put: UniFFI API checksum mismatch") + panic("slatedb: uniffi_slatedb_uniffi_checksum_method_writehandle_create_ts: UniFFI API checksum mismatch") } } { checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { - return C.uniffi_slatedb_uniffi_checksum_method_writebatch_put_with_options() + return C.uniffi_slatedb_uniffi_checksum_method_writehandle_seqnum() }) - if checksum != 31177 { + if checksum != 19654 { // If this happens try cleaning and rebuilding your project - panic("slatedb: uniffi_slatedb_uniffi_checksum_method_writebatch_put_with_options: UniFFI API checksum mismatch") + panic("slatedb: uniffi_slatedb_uniffi_checksum_method_writehandle_seqnum: UniFFI API checksum mismatch") } } { @@ -1859,11 +1850,38 @@ func uniffiCheckChecksums() { } { checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { - return C.uniffi_slatedb_uniffi_checksum_constructor_walreader_new() + return C.uniffi_slatedb_uniffi_checksum_constructor_slatedbwalreader_new() + }) + if checksum != 59531 { + // If this happens try cleaning and rebuilding your project + panic("slatedb: uniffi_slatedb_uniffi_checksum_constructor_slatedbwalreader_new: UniFFI API checksum mismatch") + } + } + { + checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { + return C.uniffi_slatedb_uniffi_checksum_constructor_slatedbwalreader_with_options() + }) + if checksum != 41949 { + // If this happens try cleaning and rebuilding your project + panic("slatedb: uniffi_slatedb_uniffi_checksum_constructor_slatedbwalreader_with_options: UniFFI API checksum mismatch") + } + } + { + checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { + return C.uniffi_slatedb_uniffi_checksum_constructor_slatedbwalreader_with_wal_object_store() + }) + if checksum != 35831 { + // If this happens try cleaning and rebuilding your project + panic("slatedb: uniffi_slatedb_uniffi_checksum_constructor_slatedbwalreader_with_wal_object_store: UniFFI API checksum mismatch") + } + } + { + checksum := rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint16_t { + return C.uniffi_slatedb_uniffi_checksum_constructor_slatedbwalreader_with_wal_object_store_and_options() }) - if checksum != 30537 { + if checksum != 28081 { // If this happens try cleaning and rebuilding your project - panic("slatedb: uniffi_slatedb_uniffi_checksum_constructor_walreader_new: UniFFI API checksum mismatch") + panic("slatedb: uniffi_slatedb_uniffi_checksum_constructor_slatedbwalreader_with_wal_object_store_and_options: UniFFI API checksum mismatch") } } { @@ -3265,9 +3283,9 @@ type DbInterface interface { // Starts a transaction at the requested isolation level. Begin(isolationLevel IsolationLevel) (*DbTransaction, error) // Deletes `key` and returns metadata for the write. - Delete(key []byte) (WriteHandle, error) + Delete(key []byte) (*WriteHandle, error) // Deletes `key` using custom write options. - DeleteWithOptions(key []byte, options WriteOptions) (WriteHandle, error) + DeleteWithOptions(key []byte, options WriteOptions) (*WriteHandle, error) // Best-effort eviction of block-cache entries for one SST. // // If no block cache is configured, returns `Ok(())`. @@ -3285,16 +3303,16 @@ type DbInterface interface { // Reads the current value for `key` using custom read options. GetWithOptions(key []byte, options ReadOptions) (*[]byte, error) // Appends a merge operand for `key` and returns metadata for the write. - Merge(key []byte, operand []byte) (WriteHandle, error) + Merge(key []byte, operand []byte) (*WriteHandle, error) // Appends a merge operand using custom merge and write options. - MergeWithOptions(key []byte, operand []byte, mergeOptions MergeOptions, writeOptions WriteOptions) (WriteHandle, error) + MergeWithOptions(key []byte, operand []byte, mergeOptions MergeOptions, writeOptions WriteOptions) (*WriteHandle, error) // Inserts or overwrites a value and returns metadata for the write. // // Keys must be non-empty and at most `u16::MAX` bytes. Values must be at // most `u32::MAX` bytes. - Put(key []byte, value []byte) (WriteHandle, error) + Put(key []byte, value []byte) (*WriteHandle, error) // Inserts or overwrites a value using custom put and write options. - PutWithOptions(key []byte, value []byte, putOptions PutOptions, writeOptions WriteOptions) (WriteHandle, error) + PutWithOptions(key []byte, value []byte, putOptions PutOptions, writeOptions WriteOptions) (*WriteHandle, error) // Scans rows inside `range`. Scan(varRange KeyRange) (*DbIterator, error) // Scans rows whose keys start with `prefix`, restricted to `subrange`. @@ -3306,6 +3324,8 @@ type DbInterface interface { ScanWithOptions(varRange KeyRange, options ScanOptions) (*DbIterator, error) // Flushes outstanding work and closes the database. Shutdown() error + // Performs the requested final flush and closes the database. + ShutdownWithOptions(options CloseOptions) error // Creates a read-only snapshot representing a consistent point in time. Snapshot() (*DbSnapshot, error) // Returns the latest database status snapshot, including the segment @@ -3320,11 +3340,11 @@ type DbInterface interface { // Applies all operations in `batch` atomically. // // The provided batch is consumed and cannot be reused afterwards. - Write(batch *WriteBatch) (WriteHandle, error) + Write(batch *WriteBatch) (*WriteHandle, error) // Applies all operations in `batch` atomically using custom write options. // // The provided batch is consumed and cannot be reused afterwards. - WriteWithOptions(batch *WriteBatch, options WriteOptions) (WriteHandle, error) + WriteWithOptions(batch *WriteBatch, options WriteOptions) (*WriteHandle, error) } // A writable SlateDB handle. @@ -3367,31 +3387,29 @@ func (_self *Db) Begin(isolationLevel IsolationLevel) (*DbTransaction, error) { } // Deletes `key` and returns metadata for the write. -func (_self *Db) Delete(key []byte) (WriteHandle, error) { +func (_self *Db) Delete(key []byte) (*WriteHandle, error) { _pointer := _self.ffiObject.incrementPointer("*Db") defer _self.ffiObject.decrementPointer() res, err := uniffiRustCallAsync[*Error]( FfiConverterErrorINSTANCE, // completeFn - func(handle C.uint64_t, status *C.RustCallStatus) RustBufferI { - res := C.ffi_slatedb_uniffi_rust_future_complete_rust_buffer(handle, status) - return GoRustBuffer{ - inner: res, - } + func(handle C.uint64_t, status *C.RustCallStatus) C.uint64_t { + res := C.ffi_slatedb_uniffi_rust_future_complete_u64(handle, status) + return res }, // liftFn - func(ffi RustBufferI) WriteHandle { + func(ffi C.uint64_t) *WriteHandle { return FfiConverterWriteHandleINSTANCE.Lift(ffi) }, C.uniffi_slatedb_uniffi_fn_method_db_delete( _pointer, FfiConverterBytesINSTANCE.Lower(key)), // pollFn func(handle C.uint64_t, continuation C.UniffiRustFutureContinuationCallback, data C.uint64_t) { - C.ffi_slatedb_uniffi_rust_future_poll_rust_buffer(handle, continuation, data) + C.ffi_slatedb_uniffi_rust_future_poll_u64(handle, continuation, data) }, // freeFn func(handle C.uint64_t) { - C.ffi_slatedb_uniffi_rust_future_free_rust_buffer(handle) + C.ffi_slatedb_uniffi_rust_future_free_u64(handle) }, ) @@ -3403,31 +3421,29 @@ func (_self *Db) Delete(key []byte) (WriteHandle, error) { } // Deletes `key` using custom write options. -func (_self *Db) DeleteWithOptions(key []byte, options WriteOptions) (WriteHandle, error) { +func (_self *Db) DeleteWithOptions(key []byte, options WriteOptions) (*WriteHandle, error) { _pointer := _self.ffiObject.incrementPointer("*Db") defer _self.ffiObject.decrementPointer() res, err := uniffiRustCallAsync[*Error]( FfiConverterErrorINSTANCE, // completeFn - func(handle C.uint64_t, status *C.RustCallStatus) RustBufferI { - res := C.ffi_slatedb_uniffi_rust_future_complete_rust_buffer(handle, status) - return GoRustBuffer{ - inner: res, - } + func(handle C.uint64_t, status *C.RustCallStatus) C.uint64_t { + res := C.ffi_slatedb_uniffi_rust_future_complete_u64(handle, status) + return res }, // liftFn - func(ffi RustBufferI) WriteHandle { + func(ffi C.uint64_t) *WriteHandle { return FfiConverterWriteHandleINSTANCE.Lift(ffi) }, C.uniffi_slatedb_uniffi_fn_method_db_delete_with_options( _pointer, FfiConverterBytesINSTANCE.Lower(key), FfiConverterWriteOptionsINSTANCE.Lower(options)), // pollFn func(handle C.uint64_t, continuation C.UniffiRustFutureContinuationCallback, data C.uint64_t) { - C.ffi_slatedb_uniffi_rust_future_poll_rust_buffer(handle, continuation, data) + C.ffi_slatedb_uniffi_rust_future_poll_u64(handle, continuation, data) }, // freeFn func(handle C.uint64_t) { - C.ffi_slatedb_uniffi_rust_future_free_rust_buffer(handle) + C.ffi_slatedb_uniffi_rust_future_free_u64(handle) }, ) @@ -3681,31 +3697,29 @@ func (_self *Db) GetWithOptions(key []byte, options ReadOptions) (*[]byte, error } // Appends a merge operand for `key` and returns metadata for the write. -func (_self *Db) Merge(key []byte, operand []byte) (WriteHandle, error) { +func (_self *Db) Merge(key []byte, operand []byte) (*WriteHandle, error) { _pointer := _self.ffiObject.incrementPointer("*Db") defer _self.ffiObject.decrementPointer() res, err := uniffiRustCallAsync[*Error]( FfiConverterErrorINSTANCE, // completeFn - func(handle C.uint64_t, status *C.RustCallStatus) RustBufferI { - res := C.ffi_slatedb_uniffi_rust_future_complete_rust_buffer(handle, status) - return GoRustBuffer{ - inner: res, - } + func(handle C.uint64_t, status *C.RustCallStatus) C.uint64_t { + res := C.ffi_slatedb_uniffi_rust_future_complete_u64(handle, status) + return res }, // liftFn - func(ffi RustBufferI) WriteHandle { + func(ffi C.uint64_t) *WriteHandle { return FfiConverterWriteHandleINSTANCE.Lift(ffi) }, C.uniffi_slatedb_uniffi_fn_method_db_merge( _pointer, FfiConverterBytesINSTANCE.Lower(key), FfiConverterBytesINSTANCE.Lower(operand)), // pollFn func(handle C.uint64_t, continuation C.UniffiRustFutureContinuationCallback, data C.uint64_t) { - C.ffi_slatedb_uniffi_rust_future_poll_rust_buffer(handle, continuation, data) + C.ffi_slatedb_uniffi_rust_future_poll_u64(handle, continuation, data) }, // freeFn func(handle C.uint64_t) { - C.ffi_slatedb_uniffi_rust_future_free_rust_buffer(handle) + C.ffi_slatedb_uniffi_rust_future_free_u64(handle) }, ) @@ -3717,31 +3731,29 @@ func (_self *Db) Merge(key []byte, operand []byte) (WriteHandle, error) { } // Appends a merge operand using custom merge and write options. -func (_self *Db) MergeWithOptions(key []byte, operand []byte, mergeOptions MergeOptions, writeOptions WriteOptions) (WriteHandle, error) { +func (_self *Db) MergeWithOptions(key []byte, operand []byte, mergeOptions MergeOptions, writeOptions WriteOptions) (*WriteHandle, error) { _pointer := _self.ffiObject.incrementPointer("*Db") defer _self.ffiObject.decrementPointer() res, err := uniffiRustCallAsync[*Error]( FfiConverterErrorINSTANCE, // completeFn - func(handle C.uint64_t, status *C.RustCallStatus) RustBufferI { - res := C.ffi_slatedb_uniffi_rust_future_complete_rust_buffer(handle, status) - return GoRustBuffer{ - inner: res, - } + func(handle C.uint64_t, status *C.RustCallStatus) C.uint64_t { + res := C.ffi_slatedb_uniffi_rust_future_complete_u64(handle, status) + return res }, // liftFn - func(ffi RustBufferI) WriteHandle { + func(ffi C.uint64_t) *WriteHandle { return FfiConverterWriteHandleINSTANCE.Lift(ffi) }, C.uniffi_slatedb_uniffi_fn_method_db_merge_with_options( _pointer, FfiConverterBytesINSTANCE.Lower(key), FfiConverterBytesINSTANCE.Lower(operand), FfiConverterMergeOptionsINSTANCE.Lower(mergeOptions), FfiConverterWriteOptionsINSTANCE.Lower(writeOptions)), // pollFn func(handle C.uint64_t, continuation C.UniffiRustFutureContinuationCallback, data C.uint64_t) { - C.ffi_slatedb_uniffi_rust_future_poll_rust_buffer(handle, continuation, data) + C.ffi_slatedb_uniffi_rust_future_poll_u64(handle, continuation, data) }, // freeFn func(handle C.uint64_t) { - C.ffi_slatedb_uniffi_rust_future_free_rust_buffer(handle) + C.ffi_slatedb_uniffi_rust_future_free_u64(handle) }, ) @@ -3756,31 +3768,29 @@ func (_self *Db) MergeWithOptions(key []byte, operand []byte, mergeOptions Merge // // Keys must be non-empty and at most `u16::MAX` bytes. Values must be at // most `u32::MAX` bytes. -func (_self *Db) Put(key []byte, value []byte) (WriteHandle, error) { +func (_self *Db) Put(key []byte, value []byte) (*WriteHandle, error) { _pointer := _self.ffiObject.incrementPointer("*Db") defer _self.ffiObject.decrementPointer() res, err := uniffiRustCallAsync[*Error]( FfiConverterErrorINSTANCE, // completeFn - func(handle C.uint64_t, status *C.RustCallStatus) RustBufferI { - res := C.ffi_slatedb_uniffi_rust_future_complete_rust_buffer(handle, status) - return GoRustBuffer{ - inner: res, - } + func(handle C.uint64_t, status *C.RustCallStatus) C.uint64_t { + res := C.ffi_slatedb_uniffi_rust_future_complete_u64(handle, status) + return res }, // liftFn - func(ffi RustBufferI) WriteHandle { + func(ffi C.uint64_t) *WriteHandle { return FfiConverterWriteHandleINSTANCE.Lift(ffi) }, C.uniffi_slatedb_uniffi_fn_method_db_put( _pointer, FfiConverterBytesINSTANCE.Lower(key), FfiConverterBytesINSTANCE.Lower(value)), // pollFn func(handle C.uint64_t, continuation C.UniffiRustFutureContinuationCallback, data C.uint64_t) { - C.ffi_slatedb_uniffi_rust_future_poll_rust_buffer(handle, continuation, data) + C.ffi_slatedb_uniffi_rust_future_poll_u64(handle, continuation, data) }, // freeFn func(handle C.uint64_t) { - C.ffi_slatedb_uniffi_rust_future_free_rust_buffer(handle) + C.ffi_slatedb_uniffi_rust_future_free_u64(handle) }, ) @@ -3792,31 +3802,29 @@ func (_self *Db) Put(key []byte, value []byte) (WriteHandle, error) { } // Inserts or overwrites a value using custom put and write options. -func (_self *Db) PutWithOptions(key []byte, value []byte, putOptions PutOptions, writeOptions WriteOptions) (WriteHandle, error) { +func (_self *Db) PutWithOptions(key []byte, value []byte, putOptions PutOptions, writeOptions WriteOptions) (*WriteHandle, error) { _pointer := _self.ffiObject.incrementPointer("*Db") defer _self.ffiObject.decrementPointer() res, err := uniffiRustCallAsync[*Error]( FfiConverterErrorINSTANCE, // completeFn - func(handle C.uint64_t, status *C.RustCallStatus) RustBufferI { - res := C.ffi_slatedb_uniffi_rust_future_complete_rust_buffer(handle, status) - return GoRustBuffer{ - inner: res, - } + func(handle C.uint64_t, status *C.RustCallStatus) C.uint64_t { + res := C.ffi_slatedb_uniffi_rust_future_complete_u64(handle, status) + return res }, // liftFn - func(ffi RustBufferI) WriteHandle { + func(ffi C.uint64_t) *WriteHandle { return FfiConverterWriteHandleINSTANCE.Lift(ffi) }, C.uniffi_slatedb_uniffi_fn_method_db_put_with_options( _pointer, FfiConverterBytesINSTANCE.Lower(key), FfiConverterBytesINSTANCE.Lower(value), FfiConverterPutOptionsINSTANCE.Lower(putOptions), FfiConverterWriteOptionsINSTANCE.Lower(writeOptions)), // pollFn func(handle C.uint64_t, continuation C.UniffiRustFutureContinuationCallback, data C.uint64_t) { - C.ffi_slatedb_uniffi_rust_future_poll_rust_buffer(handle, continuation, data) + C.ffi_slatedb_uniffi_rust_future_poll_u64(handle, continuation, data) }, // freeFn func(handle C.uint64_t) { - C.ffi_slatedb_uniffi_rust_future_free_rust_buffer(handle) + C.ffi_slatedb_uniffi_rust_future_free_u64(handle) }, ) @@ -3996,6 +4004,38 @@ func (_self *Db) Shutdown() error { return err } +// Performs the requested final flush and closes the database. +func (_self *Db) ShutdownWithOptions(options CloseOptions) error { + _pointer := _self.ffiObject.incrementPointer("*Db") + defer _self.ffiObject.decrementPointer() + _, err := uniffiRustCallAsync[*Error]( + FfiConverterErrorINSTANCE, + // completeFn + func(handle C.uint64_t, status *C.RustCallStatus) struct{} { + C.ffi_slatedb_uniffi_rust_future_complete_void(handle, status) + return struct{}{} + }, + // liftFn + func(_ struct{}) struct{} { return struct{}{} }, + C.uniffi_slatedb_uniffi_fn_method_db_shutdown_with_options( + _pointer, FfiConverterCloseOptionsINSTANCE.Lower(options)), + // pollFn + func(handle C.uint64_t, continuation C.UniffiRustFutureContinuationCallback, data C.uint64_t) { + C.ffi_slatedb_uniffi_rust_future_poll_void(handle, continuation, data) + }, + // freeFn + func(handle C.uint64_t) { + C.ffi_slatedb_uniffi_rust_future_free_void(handle) + }, + ) + + if err == nil { + return nil + } + + return err +} + // Creates a read-only snapshot representing a consistent point in time. func (_self *Db) Snapshot() (*DbSnapshot, error) { _pointer := _self.ffiObject.incrementPointer("*Db") @@ -4082,31 +4122,29 @@ func (_self *Db) WarmSst(sstId SsTableId, targets []CacheTarget) error { // Applies all operations in `batch` atomically. // // The provided batch is consumed and cannot be reused afterwards. -func (_self *Db) Write(batch *WriteBatch) (WriteHandle, error) { +func (_self *Db) Write(batch *WriteBatch) (*WriteHandle, error) { _pointer := _self.ffiObject.incrementPointer("*Db") defer _self.ffiObject.decrementPointer() res, err := uniffiRustCallAsync[*Error]( FfiConverterErrorINSTANCE, // completeFn - func(handle C.uint64_t, status *C.RustCallStatus) RustBufferI { - res := C.ffi_slatedb_uniffi_rust_future_complete_rust_buffer(handle, status) - return GoRustBuffer{ - inner: res, - } + func(handle C.uint64_t, status *C.RustCallStatus) C.uint64_t { + res := C.ffi_slatedb_uniffi_rust_future_complete_u64(handle, status) + return res }, // liftFn - func(ffi RustBufferI) WriteHandle { + func(ffi C.uint64_t) *WriteHandle { return FfiConverterWriteHandleINSTANCE.Lift(ffi) }, C.uniffi_slatedb_uniffi_fn_method_db_write( _pointer, FfiConverterWriteBatchINSTANCE.Lower(batch)), // pollFn func(handle C.uint64_t, continuation C.UniffiRustFutureContinuationCallback, data C.uint64_t) { - C.ffi_slatedb_uniffi_rust_future_poll_rust_buffer(handle, continuation, data) + C.ffi_slatedb_uniffi_rust_future_poll_u64(handle, continuation, data) }, // freeFn func(handle C.uint64_t) { - C.ffi_slatedb_uniffi_rust_future_free_rust_buffer(handle) + C.ffi_slatedb_uniffi_rust_future_free_u64(handle) }, ) @@ -4120,31 +4158,29 @@ func (_self *Db) Write(batch *WriteBatch) (WriteHandle, error) { // Applies all operations in `batch` atomically using custom write options. // // The provided batch is consumed and cannot be reused afterwards. -func (_self *Db) WriteWithOptions(batch *WriteBatch, options WriteOptions) (WriteHandle, error) { +func (_self *Db) WriteWithOptions(batch *WriteBatch, options WriteOptions) (*WriteHandle, error) { _pointer := _self.ffiObject.incrementPointer("*Db") defer _self.ffiObject.decrementPointer() res, err := uniffiRustCallAsync[*Error]( FfiConverterErrorINSTANCE, // completeFn - func(handle C.uint64_t, status *C.RustCallStatus) RustBufferI { - res := C.ffi_slatedb_uniffi_rust_future_complete_rust_buffer(handle, status) - return GoRustBuffer{ - inner: res, - } + func(handle C.uint64_t, status *C.RustCallStatus) C.uint64_t { + res := C.ffi_slatedb_uniffi_rust_future_complete_u64(handle, status) + return res }, // liftFn - func(ffi RustBufferI) WriteHandle { + func(ffi C.uint64_t) *WriteHandle { return FfiConverterWriteHandleINSTANCE.Lift(ffi) }, C.uniffi_slatedb_uniffi_fn_method_db_write_with_options( _pointer, FfiConverterWriteBatchINSTANCE.Lower(batch), FfiConverterWriteOptionsINSTANCE.Lower(options)), // pollFn func(handle C.uint64_t, continuation C.UniffiRustFutureContinuationCallback, data C.uint64_t) { - C.ffi_slatedb_uniffi_rust_future_poll_rust_buffer(handle, continuation, data) + C.ffi_slatedb_uniffi_rust_future_poll_u64(handle, continuation, data) }, // freeFn func(handle C.uint64_t) { - C.ffi_slatedb_uniffi_rust_future_free_rust_buffer(handle) + C.ffi_slatedb_uniffi_rust_future_free_u64(handle) }, ) @@ -5856,11 +5892,11 @@ type DbTransactionInterface interface { // Commits the transaction. // // Returns `None` when the transaction performed no writes. - Commit() (*WriteHandle, error) + Commit() (**WriteHandle, error) // Commits the transaction using custom write options. // // Returns `None` when the transaction performed no writes. - CommitWithOptions(options WriteOptions) (*WriteHandle, error) + CommitWithOptions(options WriteOptions) (**WriteHandle, error) // Buffers a delete inside the transaction. Delete(key []byte) error // Reads the value visible to this transaction for `key`. @@ -5912,7 +5948,7 @@ type DbTransaction struct { // Commits the transaction. // // Returns `None` when the transaction performed no writes. -func (_self *DbTransaction) Commit() (*WriteHandle, error) { +func (_self *DbTransaction) Commit() (**WriteHandle, error) { _pointer := _self.ffiObject.incrementPointer("*DbTransaction") defer _self.ffiObject.decrementPointer() res, err := uniffiRustCallAsync[*Error]( @@ -5925,7 +5961,7 @@ func (_self *DbTransaction) Commit() (*WriteHandle, error) { } }, // liftFn - func(ffi RustBufferI) *WriteHandle { + func(ffi RustBufferI) **WriteHandle { return FfiConverterOptionalWriteHandleINSTANCE.Lift(ffi) }, C.uniffi_slatedb_uniffi_fn_method_dbtransaction_commit( @@ -5950,7 +5986,7 @@ func (_self *DbTransaction) Commit() (*WriteHandle, error) { // Commits the transaction using custom write options. // // Returns `None` when the transaction performed no writes. -func (_self *DbTransaction) CommitWithOptions(options WriteOptions) (*WriteHandle, error) { +func (_self *DbTransaction) CommitWithOptions(options WriteOptions) (**WriteHandle, error) { _pointer := _self.ffiObject.incrementPointer("*DbTransaction") defer _self.ffiObject.decrementPointer() res, err := uniffiRustCallAsync[*Error]( @@ -5963,7 +5999,7 @@ func (_self *DbTransaction) CommitWithOptions(options WriteOptions) (*WriteHandl } }, // liftFn - func(ffi RustBufferI) *WriteHandle { + func(ffi RustBufferI) **WriteHandle { return FfiConverterOptionalWriteHandleINSTANCE.Lift(ffi) }, C.uniffi_slatedb_uniffi_fn_method_dbtransaction_commit_with_options( @@ -7999,8 +8035,8 @@ type SettingsInterface interface { // Examples: // // - `set("flush_interval", "\"250ms\"")` - // - `set("default_ttl", "42")` - // - `set("default_ttl", "null")` + // - `set("default_ttl_millis", "42")` + // - `set("default_ttl_millis", "null")` // - `set("compactor_options.max_sst_size", "33554432")` // - `set("object_store_cache_options.root_folder", "\"/tmp/slatedb-cache\"")` Set(key string, valueJson string) error @@ -8104,8 +8140,8 @@ func SettingsLoad() (*Settings, error) { // Examples: // // - `set("flush_interval", "\"250ms\"")` -// - `set("default_ttl", "42")` -// - `set("default_ttl", "null")` +// - `set("default_ttl_millis", "42")` +// - `set("default_ttl_millis", "null")` // - `set("compactor_options.max_sst_size", "33554432")` // - `set("object_store_cache_options.root_folder", "\"/tmp/slatedb-cache\"")` func (_self *Settings) Set(key string, valueJson string) error { @@ -8192,176 +8228,181 @@ func (_ FfiDestroyerSettings) Destroy(value *Settings) { value.Destroy() } -// Handle for an up/down counter metric. -type UpDownCounter interface { - // Adds `value` to the counter. - Increment(value int64) +// Live iterator over SlateDB WAL files starting at a required WAL file ID. +type SlateDbWalIteratorInterface interface { + // Returns rows from the next fully consumed WAL file. When it reaches the + // current tail, this call waits for the next WAL file rather than ending. + Next() (*WalRows, error) } -// Handle for an up/down counter metric. -type UpDownCounterImpl struct { +// Live iterator over SlateDB WAL files starting at a required WAL file ID. +type SlateDbWalIterator struct { ffiObject FfiObject } -// Adds `value` to the counter. -func (_self *UpDownCounterImpl) Increment(value int64) { - _pointer := _self.ffiObject.incrementPointer("UpDownCounter") +// Returns rows from the next fully consumed WAL file. When it reaches the +// current tail, this call waits for the next WAL file rather than ending. +func (_self *SlateDbWalIterator) Next() (*WalRows, error) { + _pointer := _self.ffiObject.incrementPointer("*SlateDbWalIterator") defer _self.ffiObject.decrementPointer() - rustCall(func(_uniffiStatus *C.RustCallStatus) bool { - C.uniffi_slatedb_uniffi_fn_method_updowncounter_increment( - _pointer, FfiConverterInt64INSTANCE.Lower(value), _uniffiStatus) - return false - }) + res, err := uniffiRustCallAsync[*Error]( + FfiConverterErrorINSTANCE, + // completeFn + func(handle C.uint64_t, status *C.RustCallStatus) RustBufferI { + res := C.ffi_slatedb_uniffi_rust_future_complete_rust_buffer(handle, status) + return GoRustBuffer{ + inner: res, + } + }, + // liftFn + func(ffi RustBufferI) *WalRows { + return FfiConverterOptionalWalRowsINSTANCE.Lift(ffi) + }, + C.uniffi_slatedb_uniffi_fn_method_slatedbwaliterator_next( + _pointer), + // pollFn + func(handle C.uint64_t, continuation C.UniffiRustFutureContinuationCallback, data C.uint64_t) { + C.ffi_slatedb_uniffi_rust_future_poll_rust_buffer(handle, continuation, data) + }, + // freeFn + func(handle C.uint64_t) { + C.ffi_slatedb_uniffi_rust_future_free_rust_buffer(handle) + }, + ) + + if err == nil { + return res, nil + } + + return res, err } -func (object *UpDownCounterImpl) Destroy() { +func (object *SlateDbWalIterator) Destroy() { runtime.SetFinalizer(object, nil) object.ffiObject.destroy() } -type FfiConverterUpDownCounter struct { - handleMap *concurrentHandleMap[UpDownCounter] +type FfiConverterSlateDbWalIterator struct{} + +var FfiConverterSlateDbWalIteratorINSTANCE = FfiConverterSlateDbWalIterator{} + +func (c FfiConverterSlateDbWalIterator) Lift(handle C.uint64_t) *SlateDbWalIterator { + result := &SlateDbWalIterator{ + newFfiObject( + handle, + func(handle C.uint64_t, status *C.RustCallStatus) C.uint64_t { + return C.uniffi_slatedb_uniffi_fn_clone_slatedbwaliterator(handle, status) + }, + func(handle C.uint64_t, status *C.RustCallStatus) { + C.uniffi_slatedb_uniffi_fn_free_slatedbwaliterator(handle, status) + }, + ), + } + runtime.SetFinalizer(result, (*SlateDbWalIterator).Destroy) + return result } -var FfiConverterUpDownCounterINSTANCE = FfiConverterUpDownCounter{ - handleMap: newConcurrentHandleMap[UpDownCounter](), +func (c FfiConverterSlateDbWalIterator) Read(reader io.Reader) *SlateDbWalIterator { + return c.Lift(C.uint64_t(readUint64(reader))) } -func (c FfiConverterUpDownCounter) Lift(handle C.uint64_t) UpDownCounter { - if uint64(handle)&1 == 0 { - // Rust-generated handle (even), construct a new object wrapping the handle - result := &UpDownCounterImpl{ - newFfiObject( - handle, - func(handle C.uint64_t, status *C.RustCallStatus) C.uint64_t { - return C.uniffi_slatedb_uniffi_fn_clone_updowncounter(handle, status) - }, - func(handle C.uint64_t, status *C.RustCallStatus) { - C.uniffi_slatedb_uniffi_fn_free_updowncounter(handle, status) - }, - ), - } - runtime.SetFinalizer(result, (*UpDownCounterImpl).Destroy) - return result - } else { - // Go-generated handle (odd), retrieve from the handle map - val, ok := c.handleMap.tryGet(uint64(handle)) - if !ok { - panic(fmt.Errorf("no callback in handle map: %d", handle)) - } - c.handleMap.remove(uint64(handle)) - return val - } -} - -func (c FfiConverterUpDownCounter) Read(reader io.Reader) UpDownCounter { - return c.Lift(C.uint64_t(readUint64(reader))) -} - -func (c FfiConverterUpDownCounter) Lower(value UpDownCounter) C.uint64_t { +func (c FfiConverterSlateDbWalIterator) Lower(value *SlateDbWalIterator) C.uint64_t { // TODO: this is bad - all synchronization from ObjectRuntime.go is discarded here, // because the handle will be decremented immediately after this function returns, // and someone will be left holding onto a non-locked handle. - if val, ok := value.(*UpDownCounterImpl); ok { - // Rust-backed object, clone the handle - handle := val.ffiObject.incrementPointer("UpDownCounter") - defer val.ffiObject.decrementPointer() - return handle - } else { - // Go-backed object, insert into handle map - return C.uint64_t(c.handleMap.insert(value)) - } + handle := value.ffiObject.incrementPointer("*SlateDbWalIterator") + defer value.ffiObject.decrementPointer() + return handle } -func (c FfiConverterUpDownCounter) Write(writer io.Writer, value UpDownCounter) { +func (c FfiConverterSlateDbWalIterator) Write(writer io.Writer, value *SlateDbWalIterator) { writeUint64(writer, uint64(c.Lower(value))) } -func LiftFromExternalUpDownCounter(handle uint64) UpDownCounter { - return FfiConverterUpDownCounterINSTANCE.Lift(C.uint64_t(handle)) +func LiftFromExternalSlateDbWalIterator(handle uint64) *SlateDbWalIterator { + return FfiConverterSlateDbWalIteratorINSTANCE.Lift(C.uint64_t(handle)) } -func LowerToExternalUpDownCounter(value UpDownCounter) uint64 { - return uint64(FfiConverterUpDownCounterINSTANCE.Lower(value)) +func LowerToExternalSlateDbWalIterator(value *SlateDbWalIterator) uint64 { + return uint64(FfiConverterSlateDbWalIteratorINSTANCE.Lower(value)) } -type FfiDestroyerUpDownCounter struct{} - -func (_ FfiDestroyerUpDownCounter) Destroy(value UpDownCounter) { - if val, ok := value.(*UpDownCounterImpl); ok { - val.Destroy() - } -} - -//export slatedb_uniffi_metrics_cgo_dispatchCallbackInterfaceUpDownCounterMethod0 -func slatedb_uniffi_metrics_cgo_dispatchCallbackInterfaceUpDownCounterMethod0(uniffiHandle C.uint64_t, value C.int64_t, uniffiOutReturn *C.void, callStatus *C.RustCallStatus) { - handle := uint64(uniffiHandle) - uniffiObj, ok := FfiConverterUpDownCounterINSTANCE.handleMap.tryGet(handle) - if !ok { - panic(fmt.Errorf("no callback in handle map: %d", handle)) - } - - uniffiObj.Increment( - FfiConverterInt64INSTANCE.Lift(value), - ) +type FfiDestroyerSlateDbWalIterator struct{} +func (_ FfiDestroyerSlateDbWalIterator) Destroy(value *SlateDbWalIterator) { + value.Destroy() } -var UniffiVTableCallbackInterfaceUpDownCounterINSTANCE = C.UniffiVTableCallbackInterfaceUpDownCounter{ - uniffiFree: (C.UniffiCallbackInterfaceFree)(C.slatedb_uniffi_metrics_cgo_dispatchCallbackInterfaceUpDownCounterFree), - uniffiClone: (C.UniffiCallbackInterfaceClone)(C.slatedb_uniffi_metrics_cgo_dispatchCallbackInterfaceUpDownCounterClone), - increment: (C.UniffiCallbackInterfaceUpDownCounterMethod0)(C.slatedb_uniffi_metrics_cgo_dispatchCallbackInterfaceUpDownCounterMethod0), +// CDC reader backed by SlateDB's native live WAL reader. +type SlateDbWalReaderInterface interface { + // Opens a live iterator starting at start_wal_file_id. The iterator waits + // and polls internally when it reaches the current WAL tail. + Iterator(startWalFileId uint64) (*SlateDbWalIterator, error) + // Returns a snapshot of the current WAL tail after replay_after_wal_id, or + // the supplied ID when no later WAL file exists. + LastWalFileId(replayAfterWalId uint64) (uint64, error) } -//export slatedb_uniffi_metrics_cgo_dispatchCallbackInterfaceUpDownCounterFree -func slatedb_uniffi_metrics_cgo_dispatchCallbackInterfaceUpDownCounterFree(handle C.uint64_t) { - FfiConverterUpDownCounterINSTANCE.handleMap.remove(uint64(handle)) +// CDC reader backed by SlateDB's native live WAL reader. +type SlateDbWalReader struct { + ffiObject FfiObject } -//export slatedb_uniffi_metrics_cgo_dispatchCallbackInterfaceUpDownCounterClone -func slatedb_uniffi_metrics_cgo_dispatchCallbackInterfaceUpDownCounterClone(handle C.uint64_t) C.uint64_t { - val, ok := FfiConverterUpDownCounterINSTANCE.handleMap.tryGet(uint64(handle)) - if !ok { - panic(fmt.Errorf("no callback in handle map: %d", handle)) +// Opens a reader when the manifest and WAL use the same object store. +func NewSlateDbWalReader(path string, objectStore *ObjectStore) (*SlateDbWalReader, error) { + _uniffiRV, _uniffiErr := rustCallWithError[*Error](FfiConverterError{}, func(_uniffiStatus *C.RustCallStatus) C.uint64_t { + return C.uniffi_slatedb_uniffi_fn_constructor_slatedbwalreader_new(FfiConverterStringINSTANCE.Lower(path), FfiConverterObjectStoreINSTANCE.Lower(objectStore), _uniffiStatus) + }) + if _uniffiErr != nil { + var _uniffiDefaultValue *SlateDbWalReader + return _uniffiDefaultValue, _uniffiErr + } else { + return FfiConverterSlateDbWalReaderINSTANCE.Lift(_uniffiRV), nil } - return C.uint64_t(FfiConverterUpDownCounterINSTANCE.handleMap.insert(val)) -} - -func (c FfiConverterUpDownCounter) register() { - C.uniffi_slatedb_uniffi_fn_init_callback_vtable_updowncounter(&UniffiVTableCallbackInterfaceUpDownCounterINSTANCE) } -// Handle for a single WAL file. -type WalFileInterface interface { - // Returns the WAL file ID. - Id() uint64 - // Opens an iterator over raw row entries in this WAL file. - Iterator() (*WalFileIterator, error) - // Reads object-store metadata for this WAL file. - Metadata() (IdentifiedObjectMetadata, error) - // Returns a handle for the next WAL file ID without checking existence. - NextFile() *WalFile - // Returns the WAL ID immediately after this file. - NextId() uint64 +// Opens a reader with explicit fetch options. +func SlateDbWalReaderWithOptions(path string, objectStore *ObjectStore, options SlateDbWalReaderOptions) (*SlateDbWalReader, error) { + _uniffiRV, _uniffiErr := rustCallWithError[*Error](FfiConverterError{}, func(_uniffiStatus *C.RustCallStatus) C.uint64_t { + return C.uniffi_slatedb_uniffi_fn_constructor_slatedbwalreader_with_options(FfiConverterStringINSTANCE.Lower(path), FfiConverterObjectStoreINSTANCE.Lower(objectStore), FfiConverterSlateDbWalReaderOptionsINSTANCE.Lower(options), _uniffiStatus) + }) + if _uniffiErr != nil { + var _uniffiDefaultValue *SlateDbWalReader + return _uniffiDefaultValue, _uniffiErr + } else { + return FfiConverterSlateDbWalReaderINSTANCE.Lift(_uniffiRV), nil + } } -// Handle for a single WAL file. -type WalFile struct { - ffiObject FfiObject +// Opens a reader for a database with a dedicated WAL object store. +func SlateDbWalReaderWithWalObjectStore(path string, objectStore *ObjectStore, walObjectStore *ObjectStore) (*SlateDbWalReader, error) { + _uniffiRV, _uniffiErr := rustCallWithError[*Error](FfiConverterError{}, func(_uniffiStatus *C.RustCallStatus) C.uint64_t { + return C.uniffi_slatedb_uniffi_fn_constructor_slatedbwalreader_with_wal_object_store(FfiConverterStringINSTANCE.Lower(path), FfiConverterObjectStoreINSTANCE.Lower(objectStore), FfiConverterObjectStoreINSTANCE.Lower(walObjectStore), _uniffiStatus) + }) + if _uniffiErr != nil { + var _uniffiDefaultValue *SlateDbWalReader + return _uniffiDefaultValue, _uniffiErr + } else { + return FfiConverterSlateDbWalReaderINSTANCE.Lift(_uniffiRV), nil + } } -// Returns the WAL file ID. -func (_self *WalFile) Id() uint64 { - _pointer := _self.ffiObject.incrementPointer("*WalFile") - defer _self.ffiObject.decrementPointer() - return FfiConverterUint64INSTANCE.Lift(rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint64_t { - return C.uniffi_slatedb_uniffi_fn_method_walfile_id( - _pointer, _uniffiStatus) - })) +// Opens a reader for a dedicated WAL object store with explicit options. +func SlateDbWalReaderWithWalObjectStoreAndOptions(path string, objectStore *ObjectStore, walObjectStore *ObjectStore, options SlateDbWalReaderOptions) (*SlateDbWalReader, error) { + _uniffiRV, _uniffiErr := rustCallWithError[*Error](FfiConverterError{}, func(_uniffiStatus *C.RustCallStatus) C.uint64_t { + return C.uniffi_slatedb_uniffi_fn_constructor_slatedbwalreader_with_wal_object_store_and_options(FfiConverterStringINSTANCE.Lower(path), FfiConverterObjectStoreINSTANCE.Lower(objectStore), FfiConverterObjectStoreINSTANCE.Lower(walObjectStore), FfiConverterSlateDbWalReaderOptionsINSTANCE.Lower(options), _uniffiStatus) + }) + if _uniffiErr != nil { + var _uniffiDefaultValue *SlateDbWalReader + return _uniffiDefaultValue, _uniffiErr + } else { + return FfiConverterSlateDbWalReaderINSTANCE.Lift(_uniffiRV), nil + } } -// Opens an iterator over raw row entries in this WAL file. -func (_self *WalFile) Iterator() (*WalFileIterator, error) { - _pointer := _self.ffiObject.incrementPointer("*WalFile") +// Opens a live iterator starting at start_wal_file_id. The iterator waits +// and polls internally when it reaches the current WAL tail. +func (_self *SlateDbWalReader) Iterator(startWalFileId uint64) (*SlateDbWalIterator, error) { + _pointer := _self.ffiObject.incrementPointer("*SlateDbWalReader") defer _self.ffiObject.decrementPointer() res, err := uniffiRustCallAsync[*Error]( FfiConverterErrorINSTANCE, @@ -8371,11 +8412,11 @@ func (_self *WalFile) Iterator() (*WalFileIterator, error) { return res }, // liftFn - func(ffi C.uint64_t) *WalFileIterator { - return FfiConverterWalFileIteratorINSTANCE.Lift(ffi) + func(ffi C.uint64_t) *SlateDbWalIterator { + return FfiConverterSlateDbWalIteratorINSTANCE.Lift(ffi) }, - C.uniffi_slatedb_uniffi_fn_method_walfile_iterator( - _pointer), + C.uniffi_slatedb_uniffi_fn_method_slatedbwalreader_iterator( + _pointer, FfiConverterUint64INSTANCE.Lower(startWalFileId)), // pollFn func(handle C.uint64_t, continuation C.UniffiRustFutureContinuationCallback, data C.uint64_t) { C.ffi_slatedb_uniffi_rust_future_poll_u64(handle, continuation, data) @@ -8393,32 +8434,31 @@ func (_self *WalFile) Iterator() (*WalFileIterator, error) { return res, err } -// Reads object-store metadata for this WAL file. -func (_self *WalFile) Metadata() (IdentifiedObjectMetadata, error) { - _pointer := _self.ffiObject.incrementPointer("*WalFile") +// Returns a snapshot of the current WAL tail after replay_after_wal_id, or +// the supplied ID when no later WAL file exists. +func (_self *SlateDbWalReader) LastWalFileId(replayAfterWalId uint64) (uint64, error) { + _pointer := _self.ffiObject.incrementPointer("*SlateDbWalReader") defer _self.ffiObject.decrementPointer() res, err := uniffiRustCallAsync[*Error]( FfiConverterErrorINSTANCE, // completeFn - func(handle C.uint64_t, status *C.RustCallStatus) RustBufferI { - res := C.ffi_slatedb_uniffi_rust_future_complete_rust_buffer(handle, status) - return GoRustBuffer{ - inner: res, - } + func(handle C.uint64_t, status *C.RustCallStatus) C.uint64_t { + res := C.ffi_slatedb_uniffi_rust_future_complete_u64(handle, status) + return res }, // liftFn - func(ffi RustBufferI) IdentifiedObjectMetadata { - return FfiConverterIdentifiedObjectMetadataINSTANCE.Lift(ffi) + func(ffi C.uint64_t) uint64 { + return FfiConverterUint64INSTANCE.Lift(ffi) }, - C.uniffi_slatedb_uniffi_fn_method_walfile_metadata( - _pointer), + C.uniffi_slatedb_uniffi_fn_method_slatedbwalreader_last_wal_file_id( + _pointer, FfiConverterUint64INSTANCE.Lower(replayAfterWalId)), // pollFn func(handle C.uint64_t, continuation C.UniffiRustFutureContinuationCallback, data C.uint64_t) { - C.ffi_slatedb_uniffi_rust_future_poll_rust_buffer(handle, continuation, data) + C.ffi_slatedb_uniffi_rust_future_poll_u64(handle, continuation, data) }, // freeFn func(handle C.uint64_t) { - C.ffi_slatedb_uniffi_rust_future_free_rust_buffer(handle) + C.ffi_slatedb_uniffi_rust_future_free_u64(handle) }, ) @@ -8428,307 +8468,198 @@ func (_self *WalFile) Metadata() (IdentifiedObjectMetadata, error) { return res, err } - -// Returns a handle for the next WAL file ID without checking existence. -func (_self *WalFile) NextFile() *WalFile { - _pointer := _self.ffiObject.incrementPointer("*WalFile") - defer _self.ffiObject.decrementPointer() - return FfiConverterWalFileINSTANCE.Lift(rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint64_t { - return C.uniffi_slatedb_uniffi_fn_method_walfile_next_file( - _pointer, _uniffiStatus) - })) -} - -// Returns the WAL ID immediately after this file. -func (_self *WalFile) NextId() uint64 { - _pointer := _self.ffiObject.incrementPointer("*WalFile") - defer _self.ffiObject.decrementPointer() - return FfiConverterUint64INSTANCE.Lift(rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint64_t { - return C.uniffi_slatedb_uniffi_fn_method_walfile_next_id( - _pointer, _uniffiStatus) - })) -} -func (object *WalFile) Destroy() { +func (object *SlateDbWalReader) Destroy() { runtime.SetFinalizer(object, nil) object.ffiObject.destroy() } -type FfiConverterWalFile struct{} +type FfiConverterSlateDbWalReader struct{} -var FfiConverterWalFileINSTANCE = FfiConverterWalFile{} +var FfiConverterSlateDbWalReaderINSTANCE = FfiConverterSlateDbWalReader{} -func (c FfiConverterWalFile) Lift(handle C.uint64_t) *WalFile { - result := &WalFile{ +func (c FfiConverterSlateDbWalReader) Lift(handle C.uint64_t) *SlateDbWalReader { + result := &SlateDbWalReader{ newFfiObject( handle, func(handle C.uint64_t, status *C.RustCallStatus) C.uint64_t { - return C.uniffi_slatedb_uniffi_fn_clone_walfile(handle, status) + return C.uniffi_slatedb_uniffi_fn_clone_slatedbwalreader(handle, status) }, func(handle C.uint64_t, status *C.RustCallStatus) { - C.uniffi_slatedb_uniffi_fn_free_walfile(handle, status) + C.uniffi_slatedb_uniffi_fn_free_slatedbwalreader(handle, status) }, ), } - runtime.SetFinalizer(result, (*WalFile).Destroy) + runtime.SetFinalizer(result, (*SlateDbWalReader).Destroy) return result } -func (c FfiConverterWalFile) Read(reader io.Reader) *WalFile { +func (c FfiConverterSlateDbWalReader) Read(reader io.Reader) *SlateDbWalReader { return c.Lift(C.uint64_t(readUint64(reader))) } -func (c FfiConverterWalFile) Lower(value *WalFile) C.uint64_t { +func (c FfiConverterSlateDbWalReader) Lower(value *SlateDbWalReader) C.uint64_t { // TODO: this is bad - all synchronization from ObjectRuntime.go is discarded here, // because the handle will be decremented immediately after this function returns, // and someone will be left holding onto a non-locked handle. - handle := value.ffiObject.incrementPointer("*WalFile") + handle := value.ffiObject.incrementPointer("*SlateDbWalReader") defer value.ffiObject.decrementPointer() return handle } -func (c FfiConverterWalFile) Write(writer io.Writer, value *WalFile) { +func (c FfiConverterSlateDbWalReader) Write(writer io.Writer, value *SlateDbWalReader) { writeUint64(writer, uint64(c.Lower(value))) } -func LiftFromExternalWalFile(handle uint64) *WalFile { - return FfiConverterWalFileINSTANCE.Lift(C.uint64_t(handle)) +func LiftFromExternalSlateDbWalReader(handle uint64) *SlateDbWalReader { + return FfiConverterSlateDbWalReaderINSTANCE.Lift(C.uint64_t(handle)) } -func LowerToExternalWalFile(value *WalFile) uint64 { - return uint64(FfiConverterWalFileINSTANCE.Lower(value)) +func LowerToExternalSlateDbWalReader(value *SlateDbWalReader) uint64 { + return uint64(FfiConverterSlateDbWalReaderINSTANCE.Lower(value)) } -type FfiDestroyerWalFile struct{} +type FfiDestroyerSlateDbWalReader struct{} -func (_ FfiDestroyerWalFile) Destroy(value *WalFile) { +func (_ FfiDestroyerSlateDbWalReader) Destroy(value *SlateDbWalReader) { value.Destroy() } -// Iterator over raw row entries stored in a WAL file. -type WalFileIteratorInterface interface { - // Returns the next raw row entry from the WAL file. - Next() (*RowEntry, error) +// Handle for an up/down counter metric. +type UpDownCounter interface { + // Adds `value` to the counter. + Increment(value int64) } -// Iterator over raw row entries stored in a WAL file. -type WalFileIterator struct { +// Handle for an up/down counter metric. +type UpDownCounterImpl struct { ffiObject FfiObject } -// Returns the next raw row entry from the WAL file. -func (_self *WalFileIterator) Next() (*RowEntry, error) { - _pointer := _self.ffiObject.incrementPointer("*WalFileIterator") +// Adds `value` to the counter. +func (_self *UpDownCounterImpl) Increment(value int64) { + _pointer := _self.ffiObject.incrementPointer("UpDownCounter") defer _self.ffiObject.decrementPointer() - res, err := uniffiRustCallAsync[*Error]( - FfiConverterErrorINSTANCE, - // completeFn - func(handle C.uint64_t, status *C.RustCallStatus) RustBufferI { - res := C.ffi_slatedb_uniffi_rust_future_complete_rust_buffer(handle, status) - return GoRustBuffer{ - inner: res, - } - }, - // liftFn - func(ffi RustBufferI) *RowEntry { - return FfiConverterOptionalRowEntryINSTANCE.Lift(ffi) - }, - C.uniffi_slatedb_uniffi_fn_method_walfileiterator_next( - _pointer), - // pollFn - func(handle C.uint64_t, continuation C.UniffiRustFutureContinuationCallback, data C.uint64_t) { - C.ffi_slatedb_uniffi_rust_future_poll_rust_buffer(handle, continuation, data) - }, - // freeFn - func(handle C.uint64_t) { - C.ffi_slatedb_uniffi_rust_future_free_rust_buffer(handle) - }, - ) - - if err == nil { - return res, nil - } - - return res, err + rustCall(func(_uniffiStatus *C.RustCallStatus) bool { + C.uniffi_slatedb_uniffi_fn_method_updowncounter_increment( + _pointer, FfiConverterInt64INSTANCE.Lower(value), _uniffiStatus) + return false + }) } -func (object *WalFileIterator) Destroy() { +func (object *UpDownCounterImpl) Destroy() { runtime.SetFinalizer(object, nil) object.ffiObject.destroy() } -type FfiConverterWalFileIterator struct{} +type FfiConverterUpDownCounter struct { + handleMap *concurrentHandleMap[UpDownCounter] +} -var FfiConverterWalFileIteratorINSTANCE = FfiConverterWalFileIterator{} +var FfiConverterUpDownCounterINSTANCE = FfiConverterUpDownCounter{ + handleMap: newConcurrentHandleMap[UpDownCounter](), +} -func (c FfiConverterWalFileIterator) Lift(handle C.uint64_t) *WalFileIterator { - result := &WalFileIterator{ - newFfiObject( - handle, - func(handle C.uint64_t, status *C.RustCallStatus) C.uint64_t { - return C.uniffi_slatedb_uniffi_fn_clone_walfileiterator(handle, status) - }, - func(handle C.uint64_t, status *C.RustCallStatus) { - C.uniffi_slatedb_uniffi_fn_free_walfileiterator(handle, status) - }, - ), +func (c FfiConverterUpDownCounter) Lift(handle C.uint64_t) UpDownCounter { + if uint64(handle)&1 == 0 { + // Rust-generated handle (even), construct a new object wrapping the handle + result := &UpDownCounterImpl{ + newFfiObject( + handle, + func(handle C.uint64_t, status *C.RustCallStatus) C.uint64_t { + return C.uniffi_slatedb_uniffi_fn_clone_updowncounter(handle, status) + }, + func(handle C.uint64_t, status *C.RustCallStatus) { + C.uniffi_slatedb_uniffi_fn_free_updowncounter(handle, status) + }, + ), + } + runtime.SetFinalizer(result, (*UpDownCounterImpl).Destroy) + return result + } else { + // Go-generated handle (odd), retrieve from the handle map + val, ok := c.handleMap.tryGet(uint64(handle)) + if !ok { + panic(fmt.Errorf("no callback in handle map: %d", handle)) + } + c.handleMap.remove(uint64(handle)) + return val } - runtime.SetFinalizer(result, (*WalFileIterator).Destroy) - return result } -func (c FfiConverterWalFileIterator) Read(reader io.Reader) *WalFileIterator { +func (c FfiConverterUpDownCounter) Read(reader io.Reader) UpDownCounter { return c.Lift(C.uint64_t(readUint64(reader))) } -func (c FfiConverterWalFileIterator) Lower(value *WalFileIterator) C.uint64_t { +func (c FfiConverterUpDownCounter) Lower(value UpDownCounter) C.uint64_t { // TODO: this is bad - all synchronization from ObjectRuntime.go is discarded here, // because the handle will be decremented immediately after this function returns, // and someone will be left holding onto a non-locked handle. - handle := value.ffiObject.incrementPointer("*WalFileIterator") - defer value.ffiObject.decrementPointer() - return handle + if val, ok := value.(*UpDownCounterImpl); ok { + // Rust-backed object, clone the handle + handle := val.ffiObject.incrementPointer("UpDownCounter") + defer val.ffiObject.decrementPointer() + return handle + } else { + // Go-backed object, insert into handle map + return C.uint64_t(c.handleMap.insert(value)) + } } -func (c FfiConverterWalFileIterator) Write(writer io.Writer, value *WalFileIterator) { +func (c FfiConverterUpDownCounter) Write(writer io.Writer, value UpDownCounter) { writeUint64(writer, uint64(c.Lower(value))) } -func LiftFromExternalWalFileIterator(handle uint64) *WalFileIterator { - return FfiConverterWalFileIteratorINSTANCE.Lift(C.uint64_t(handle)) -} - -func LowerToExternalWalFileIterator(value *WalFileIterator) uint64 { - return uint64(FfiConverterWalFileIteratorINSTANCE.Lower(value)) -} - -type FfiDestroyerWalFileIterator struct{} - -func (_ FfiDestroyerWalFileIterator) Destroy(value *WalFileIterator) { - value.Destroy() -} - -// Reader for WAL files stored under a database path. -type WalReaderInterface interface { - // Returns a handle for the WAL file with the given ID. - Get(id uint64) *WalFile - // Lists WAL files in ascending ID order. - // - // `start_id` is inclusive and `end_id` is exclusive when provided. - List(startId *uint64, endId *uint64) ([]*WalFile, error) -} - -// Reader for WAL files stored under a database path. -type WalReader struct { - ffiObject FfiObject -} - -// Creates a WAL reader for `path` in `object_store`. -func NewWalReader(path string, objectStore *ObjectStore) *WalReader { - return FfiConverterWalReaderINSTANCE.Lift(rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint64_t { - return C.uniffi_slatedb_uniffi_fn_constructor_walreader_new(FfiConverterStringINSTANCE.Lower(path), FfiConverterObjectStoreINSTANCE.Lower(objectStore), _uniffiStatus) - })) +func LiftFromExternalUpDownCounter(handle uint64) UpDownCounter { + return FfiConverterUpDownCounterINSTANCE.Lift(C.uint64_t(handle)) } -// Returns a handle for the WAL file with the given ID. -func (_self *WalReader) Get(id uint64) *WalFile { - _pointer := _self.ffiObject.incrementPointer("*WalReader") - defer _self.ffiObject.decrementPointer() - return FfiConverterWalFileINSTANCE.Lift(rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint64_t { - return C.uniffi_slatedb_uniffi_fn_method_walreader_get( - _pointer, FfiConverterUint64INSTANCE.Lower(id), _uniffiStatus) - })) +func LowerToExternalUpDownCounter(value UpDownCounter) uint64 { + return uint64(FfiConverterUpDownCounterINSTANCE.Lower(value)) } -// Lists WAL files in ascending ID order. -// -// `start_id` is inclusive and `end_id` is exclusive when provided. -func (_self *WalReader) List(startId *uint64, endId *uint64) ([]*WalFile, error) { - _pointer := _self.ffiObject.incrementPointer("*WalReader") - defer _self.ffiObject.decrementPointer() - res, err := uniffiRustCallAsync[*Error]( - FfiConverterErrorINSTANCE, - // completeFn - func(handle C.uint64_t, status *C.RustCallStatus) RustBufferI { - res := C.ffi_slatedb_uniffi_rust_future_complete_rust_buffer(handle, status) - return GoRustBuffer{ - inner: res, - } - }, - // liftFn - func(ffi RustBufferI) []*WalFile { - return FfiConverterSequenceWalFileINSTANCE.Lift(ffi) - }, - C.uniffi_slatedb_uniffi_fn_method_walreader_list( - _pointer, FfiConverterOptionalUint64INSTANCE.Lower(startId), FfiConverterOptionalUint64INSTANCE.Lower(endId)), - // pollFn - func(handle C.uint64_t, continuation C.UniffiRustFutureContinuationCallback, data C.uint64_t) { - C.ffi_slatedb_uniffi_rust_future_poll_rust_buffer(handle, continuation, data) - }, - // freeFn - func(handle C.uint64_t) { - C.ffi_slatedb_uniffi_rust_future_free_rust_buffer(handle) - }, - ) +type FfiDestroyerUpDownCounter struct{} - if err == nil { - return res, nil +func (_ FfiDestroyerUpDownCounter) Destroy(value UpDownCounter) { + if val, ok := value.(*UpDownCounterImpl); ok { + val.Destroy() } - - return res, err -} -func (object *WalReader) Destroy() { - runtime.SetFinalizer(object, nil) - object.ffiObject.destroy() } -type FfiConverterWalReader struct{} - -var FfiConverterWalReaderINSTANCE = FfiConverterWalReader{} - -func (c FfiConverterWalReader) Lift(handle C.uint64_t) *WalReader { - result := &WalReader{ - newFfiObject( - handle, - func(handle C.uint64_t, status *C.RustCallStatus) C.uint64_t { - return C.uniffi_slatedb_uniffi_fn_clone_walreader(handle, status) - }, - func(handle C.uint64_t, status *C.RustCallStatus) { - C.uniffi_slatedb_uniffi_fn_free_walreader(handle, status) - }, - ), +//export slatedb_uniffi_metrics_cgo_dispatchCallbackInterfaceUpDownCounterMethod0 +func slatedb_uniffi_metrics_cgo_dispatchCallbackInterfaceUpDownCounterMethod0(uniffiHandle C.uint64_t, value C.int64_t, uniffiOutReturn *C.void, callStatus *C.RustCallStatus) { + handle := uint64(uniffiHandle) + uniffiObj, ok := FfiConverterUpDownCounterINSTANCE.handleMap.tryGet(handle) + if !ok { + panic(fmt.Errorf("no callback in handle map: %d", handle)) } - runtime.SetFinalizer(result, (*WalReader).Destroy) - return result -} -func (c FfiConverterWalReader) Read(reader io.Reader) *WalReader { - return c.Lift(C.uint64_t(readUint64(reader))) -} + uniffiObj.Increment( + FfiConverterInt64INSTANCE.Lift(value), + ) -func (c FfiConverterWalReader) Lower(value *WalReader) C.uint64_t { - // TODO: this is bad - all synchronization from ObjectRuntime.go is discarded here, - // because the handle will be decremented immediately after this function returns, - // and someone will be left holding onto a non-locked handle. - handle := value.ffiObject.incrementPointer("*WalReader") - defer value.ffiObject.decrementPointer() - return handle } -func (c FfiConverterWalReader) Write(writer io.Writer, value *WalReader) { - writeUint64(writer, uint64(c.Lower(value))) +var UniffiVTableCallbackInterfaceUpDownCounterINSTANCE = C.UniffiVTableCallbackInterfaceUpDownCounter{ + uniffiFree: (C.UniffiCallbackInterfaceFree)(C.slatedb_uniffi_metrics_cgo_dispatchCallbackInterfaceUpDownCounterFree), + uniffiClone: (C.UniffiCallbackInterfaceClone)(C.slatedb_uniffi_metrics_cgo_dispatchCallbackInterfaceUpDownCounterClone), + increment: (C.UniffiCallbackInterfaceUpDownCounterMethod0)(C.slatedb_uniffi_metrics_cgo_dispatchCallbackInterfaceUpDownCounterMethod0), } -func LiftFromExternalWalReader(handle uint64) *WalReader { - return FfiConverterWalReaderINSTANCE.Lift(C.uint64_t(handle)) +//export slatedb_uniffi_metrics_cgo_dispatchCallbackInterfaceUpDownCounterFree +func slatedb_uniffi_metrics_cgo_dispatchCallbackInterfaceUpDownCounterFree(handle C.uint64_t) { + FfiConverterUpDownCounterINSTANCE.handleMap.remove(uint64(handle)) } -func LowerToExternalWalReader(value *WalReader) uint64 { - return uint64(FfiConverterWalReaderINSTANCE.Lower(value)) +//export slatedb_uniffi_metrics_cgo_dispatchCallbackInterfaceUpDownCounterClone +func slatedb_uniffi_metrics_cgo_dispatchCallbackInterfaceUpDownCounterClone(handle C.uint64_t) C.uint64_t { + val, ok := FfiConverterUpDownCounterINSTANCE.handleMap.tryGet(uint64(handle)) + if !ok { + panic(fmt.Errorf("no callback in handle map: %d", handle)) + } + return C.uint64_t(FfiConverterUpDownCounterINSTANCE.handleMap.insert(val)) } -type FfiDestroyerWalReader struct{} - -func (_ FfiDestroyerWalReader) Destroy(value *WalReader) { - value.Destroy() +func (c FfiConverterUpDownCounter) register() { + C.uniffi_slatedb_uniffi_fn_init_callback_vtable_updowncounter(&UniffiVTableCallbackInterfaceUpDownCounterINSTANCE) } // Mutable batch of write operations applied atomically by [`crate::Db::write`]. @@ -8801,78 +8732,200 @@ func (_self *WriteBatch) MergeWithOptions(key []byte, operand []byte, options Me func (_self *WriteBatch) Put(key []byte, value []byte) error { _pointer := _self.ffiObject.incrementPointer("*WriteBatch") defer _self.ffiObject.decrementPointer() - _, _uniffiErr := rustCallWithError[*Error](FfiConverterError{}, func(_uniffiStatus *C.RustCallStatus) bool { - C.uniffi_slatedb_uniffi_fn_method_writebatch_put( - _pointer, FfiConverterBytesINSTANCE.Lower(key), FfiConverterBytesINSTANCE.Lower(value), _uniffiStatus) - return false - }) - return _uniffiErr.AsError() + _, _uniffiErr := rustCallWithError[*Error](FfiConverterError{}, func(_uniffiStatus *C.RustCallStatus) bool { + C.uniffi_slatedb_uniffi_fn_method_writebatch_put( + _pointer, FfiConverterBytesINSTANCE.Lower(key), FfiConverterBytesINSTANCE.Lower(value), _uniffiStatus) + return false + }) + return _uniffiErr.AsError() +} + +// Appends a put operation with custom put options. +func (_self *WriteBatch) PutWithOptions(key []byte, value []byte, options PutOptions) error { + _pointer := _self.ffiObject.incrementPointer("*WriteBatch") + defer _self.ffiObject.decrementPointer() + _, _uniffiErr := rustCallWithError[*Error](FfiConverterError{}, func(_uniffiStatus *C.RustCallStatus) bool { + C.uniffi_slatedb_uniffi_fn_method_writebatch_put_with_options( + _pointer, FfiConverterBytesINSTANCE.Lower(key), FfiConverterBytesINSTANCE.Lower(value), FfiConverterPutOptionsINSTANCE.Lower(options), _uniffiStatus) + return false + }) + return _uniffiErr.AsError() +} +func (object *WriteBatch) Destroy() { + runtime.SetFinalizer(object, nil) + object.ffiObject.destroy() +} + +type FfiConverterWriteBatch struct{} + +var FfiConverterWriteBatchINSTANCE = FfiConverterWriteBatch{} + +func (c FfiConverterWriteBatch) Lift(handle C.uint64_t) *WriteBatch { + result := &WriteBatch{ + newFfiObject( + handle, + func(handle C.uint64_t, status *C.RustCallStatus) C.uint64_t { + return C.uniffi_slatedb_uniffi_fn_clone_writebatch(handle, status) + }, + func(handle C.uint64_t, status *C.RustCallStatus) { + C.uniffi_slatedb_uniffi_fn_free_writebatch(handle, status) + }, + ), + } + runtime.SetFinalizer(result, (*WriteBatch).Destroy) + return result +} + +func (c FfiConverterWriteBatch) Read(reader io.Reader) *WriteBatch { + return c.Lift(C.uint64_t(readUint64(reader))) +} + +func (c FfiConverterWriteBatch) Lower(value *WriteBatch) C.uint64_t { + // TODO: this is bad - all synchronization from ObjectRuntime.go is discarded here, + // because the handle will be decremented immediately after this function returns, + // and someone will be left holding onto a non-locked handle. + handle := value.ffiObject.incrementPointer("*WriteBatch") + defer value.ffiObject.decrementPointer() + return handle +} + +func (c FfiConverterWriteBatch) Write(writer io.Writer, value *WriteBatch) { + writeUint64(writer, uint64(c.Lower(value))) +} + +func LiftFromExternalWriteBatch(handle uint64) *WriteBatch { + return FfiConverterWriteBatchINSTANCE.Lift(C.uint64_t(handle)) +} + +func LowerToExternalWriteBatch(value *WriteBatch) uint64 { + return uint64(FfiConverterWriteBatchINSTANCE.Lower(value)) +} + +type FfiDestroyerWriteBatch struct{} + +func (_ FfiDestroyerWriteBatch) Destroy(value *WriteBatch) { + value.Destroy() +} + +// Handle returned by a successful write. +type WriteHandleInterface interface { + // Waits until the write has been durably persisted. + AwaitDurable() error + // Returns the creation timestamp assigned to the write. + CreateTs() int64 + // Returns the sequence number assigned to the write. + Seqnum() uint64 +} + +// Handle returned by a successful write. +type WriteHandle struct { + ffiObject FfiObject +} + +// Waits until the write has been durably persisted. +func (_self *WriteHandle) AwaitDurable() error { + _pointer := _self.ffiObject.incrementPointer("*WriteHandle") + defer _self.ffiObject.decrementPointer() + _, err := uniffiRustCallAsync[*Error]( + FfiConverterErrorINSTANCE, + // completeFn + func(handle C.uint64_t, status *C.RustCallStatus) struct{} { + C.ffi_slatedb_uniffi_rust_future_complete_void(handle, status) + return struct{}{} + }, + // liftFn + func(_ struct{}) struct{} { return struct{}{} }, + C.uniffi_slatedb_uniffi_fn_method_writehandle_await_durable( + _pointer), + // pollFn + func(handle C.uint64_t, continuation C.UniffiRustFutureContinuationCallback, data C.uint64_t) { + C.ffi_slatedb_uniffi_rust_future_poll_void(handle, continuation, data) + }, + // freeFn + func(handle C.uint64_t) { + C.ffi_slatedb_uniffi_rust_future_free_void(handle) + }, + ) + + if err == nil { + return nil + } + + return err +} + +// Returns the creation timestamp assigned to the write. +func (_self *WriteHandle) CreateTs() int64 { + _pointer := _self.ffiObject.incrementPointer("*WriteHandle") + defer _self.ffiObject.decrementPointer() + return FfiConverterInt64INSTANCE.Lift(rustCall(func(_uniffiStatus *C.RustCallStatus) C.int64_t { + return C.uniffi_slatedb_uniffi_fn_method_writehandle_create_ts( + _pointer, _uniffiStatus) + })) } -// Appends a put operation with custom put options. -func (_self *WriteBatch) PutWithOptions(key []byte, value []byte, options PutOptions) error { - _pointer := _self.ffiObject.incrementPointer("*WriteBatch") +// Returns the sequence number assigned to the write. +func (_self *WriteHandle) Seqnum() uint64 { + _pointer := _self.ffiObject.incrementPointer("*WriteHandle") defer _self.ffiObject.decrementPointer() - _, _uniffiErr := rustCallWithError[*Error](FfiConverterError{}, func(_uniffiStatus *C.RustCallStatus) bool { - C.uniffi_slatedb_uniffi_fn_method_writebatch_put_with_options( - _pointer, FfiConverterBytesINSTANCE.Lower(key), FfiConverterBytesINSTANCE.Lower(value), FfiConverterPutOptionsINSTANCE.Lower(options), _uniffiStatus) - return false - }) - return _uniffiErr.AsError() + return FfiConverterUint64INSTANCE.Lift(rustCall(func(_uniffiStatus *C.RustCallStatus) C.uint64_t { + return C.uniffi_slatedb_uniffi_fn_method_writehandle_seqnum( + _pointer, _uniffiStatus) + })) } -func (object *WriteBatch) Destroy() { +func (object *WriteHandle) Destroy() { runtime.SetFinalizer(object, nil) object.ffiObject.destroy() } -type FfiConverterWriteBatch struct{} +type FfiConverterWriteHandle struct{} -var FfiConverterWriteBatchINSTANCE = FfiConverterWriteBatch{} +var FfiConverterWriteHandleINSTANCE = FfiConverterWriteHandle{} -func (c FfiConverterWriteBatch) Lift(handle C.uint64_t) *WriteBatch { - result := &WriteBatch{ +func (c FfiConverterWriteHandle) Lift(handle C.uint64_t) *WriteHandle { + result := &WriteHandle{ newFfiObject( handle, func(handle C.uint64_t, status *C.RustCallStatus) C.uint64_t { - return C.uniffi_slatedb_uniffi_fn_clone_writebatch(handle, status) + return C.uniffi_slatedb_uniffi_fn_clone_writehandle(handle, status) }, func(handle C.uint64_t, status *C.RustCallStatus) { - C.uniffi_slatedb_uniffi_fn_free_writebatch(handle, status) + C.uniffi_slatedb_uniffi_fn_free_writehandle(handle, status) }, ), } - runtime.SetFinalizer(result, (*WriteBatch).Destroy) + runtime.SetFinalizer(result, (*WriteHandle).Destroy) return result } -func (c FfiConverterWriteBatch) Read(reader io.Reader) *WriteBatch { +func (c FfiConverterWriteHandle) Read(reader io.Reader) *WriteHandle { return c.Lift(C.uint64_t(readUint64(reader))) } -func (c FfiConverterWriteBatch) Lower(value *WriteBatch) C.uint64_t { +func (c FfiConverterWriteHandle) Lower(value *WriteHandle) C.uint64_t { // TODO: this is bad - all synchronization from ObjectRuntime.go is discarded here, // because the handle will be decremented immediately after this function returns, // and someone will be left holding onto a non-locked handle. - handle := value.ffiObject.incrementPointer("*WriteBatch") + handle := value.ffiObject.incrementPointer("*WriteHandle") defer value.ffiObject.decrementPointer() return handle } -func (c FfiConverterWriteBatch) Write(writer io.Writer, value *WriteBatch) { +func (c FfiConverterWriteHandle) Write(writer io.Writer, value *WriteHandle) { writeUint64(writer, uint64(c.Lower(value))) } -func LiftFromExternalWriteBatch(handle uint64) *WriteBatch { - return FfiConverterWriteBatchINSTANCE.Lift(C.uint64_t(handle)) +func LiftFromExternalWriteHandle(handle uint64) *WriteHandle { + return FfiConverterWriteHandleINSTANCE.Lift(C.uint64_t(handle)) } -func LowerToExternalWriteBatch(value *WriteBatch) uint64 { - return uint64(FfiConverterWriteBatchINSTANCE.Lower(value)) +func LowerToExternalWriteHandle(value *WriteHandle) uint64 { + return uint64(FfiConverterWriteHandleINSTANCE.Lower(value)) } -type FfiDestroyerWriteBatch struct{} +type FfiDestroyerWriteHandle struct{} -func (_ FfiDestroyerWriteBatch) Destroy(value *WriteBatch) { +func (_ FfiDestroyerWriteHandle) Destroy(value *WriteHandle) { value.Destroy() } @@ -9144,6 +9197,49 @@ func (_ FfiDestroyerCloneSourceSpec) Destroy(value CloneSourceSpec) { value.Destroy() } +// Options controlling how a database is shut down. +type CloseOptions struct { + // The final flush to perform before shutdown. When `None`, no final flush is + // triggered and writes that are not durable may be lost. + FlushType *FlushType +} + +func (r *CloseOptions) Destroy() { + FfiDestroyerOptionalFlushType{}.Destroy(r.FlushType) +} + +type FfiConverterCloseOptions struct{} + +var FfiConverterCloseOptionsINSTANCE = FfiConverterCloseOptions{} + +func (c FfiConverterCloseOptions) Lift(rb RustBufferI) CloseOptions { + return LiftFromRustBuffer[CloseOptions](c, rb) +} + +func (c FfiConverterCloseOptions) Read(reader io.Reader) CloseOptions { + return CloseOptions{ + FfiConverterOptionalFlushTypeINSTANCE.Read(reader), + } +} + +func (c FfiConverterCloseOptions) Lower(value CloseOptions) C.RustBuffer { + return LowerIntoRustBuffer[CloseOptions](c, value) +} + +func (c FfiConverterCloseOptions) LowerExternal(value CloseOptions) ExternalCRustBuffer { + return RustBufferFromC(LowerIntoRustBuffer[CloseOptions](c, value)) +} + +func (c FfiConverterCloseOptions) Write(writer io.Writer, value CloseOptions) { + FfiConverterOptionalFlushTypeINSTANCE.Write(writer, value.FlushType) +} + +type FfiDestroyerCloseOptions struct{} + +func (_ FfiDestroyerCloseOptions) Destroy(value CloseOptions) { + value.Destroy() +} + // Canonical compaction record. type Compaction struct { // Compaction ULID string. @@ -10258,6 +10354,8 @@ type ReadOptions struct { // Optional context forwarded to custom filter policies; ignored by // built-in filters. FilterContext *FilterContext + // Optional caller-supplied tracing settings. + TracingOptions *TracingOptions } func (r *ReadOptions) Destroy() { @@ -10265,6 +10363,7 @@ func (r *ReadOptions) Destroy() { FfiDestroyerBool{}.Destroy(r.Dirty) FfiDestroyerBool{}.Destroy(r.CacheBlocks) FfiDestroyerOptionalFilterContext{}.Destroy(r.FilterContext) + FfiDestroyerOptionalTracingOptions{}.Destroy(r.TracingOptions) } type FfiConverterReadOptions struct{} @@ -10281,6 +10380,7 @@ func (c FfiConverterReadOptions) Read(reader io.Reader) ReadOptions { FfiConverterBoolINSTANCE.Read(reader), FfiConverterBoolINSTANCE.Read(reader), FfiConverterOptionalFilterContextINSTANCE.Read(reader), + FfiConverterOptionalTracingOptionsINSTANCE.Read(reader), } } @@ -10297,6 +10397,7 @@ func (c FfiConverterReadOptions) Write(writer io.Writer, value ReadOptions) { FfiConverterBoolINSTANCE.Write(writer, value.Dirty) FfiConverterBoolINSTANCE.Write(writer, value.CacheBlocks) FfiConverterOptionalFilterContextINSTANCE.Write(writer, value.FilterContext) + FfiConverterOptionalTracingOptionsINSTANCE.Write(writer, value.TracingOptions) } type FfiDestroyerReadOptions struct{} @@ -10309,6 +10410,8 @@ func (_ FfiDestroyerReadOptions) Destroy(value ReadOptions) { type ReaderOptions struct { // How often the reader polls for new manifests and WAL data, in milliseconds. ManifestPollIntervalMs uint64 + // How frequently an open reader probes the exact next WAL ID. + WalPollIntervalMs uint64 // Lifetime of an internally managed checkpoint, in milliseconds. CheckpointLifetimeMs uint64 // Maximum size of one in-memory table used while replaying WAL data. @@ -10324,6 +10427,7 @@ type ReaderOptions struct { func (r *ReaderOptions) Destroy() { FfiDestroyerUint64{}.Destroy(r.ManifestPollIntervalMs) + FfiDestroyerUint64{}.Destroy(r.WalPollIntervalMs) FfiDestroyerUint64{}.Destroy(r.CheckpointLifetimeMs) FfiDestroyerUint64{}.Destroy(r.MaxMemtableBytes) FfiDestroyerBool{}.Destroy(r.SkipWalReplay) @@ -10343,6 +10447,7 @@ func (c FfiConverterReaderOptions) Read(reader io.Reader) ReaderOptions { FfiConverterUint64INSTANCE.Read(reader), FfiConverterUint64INSTANCE.Read(reader), FfiConverterUint64INSTANCE.Read(reader), + FfiConverterUint64INSTANCE.Read(reader), FfiConverterBoolINSTANCE.Read(reader), FfiConverterOptionalUint32INSTANCE.Read(reader), } @@ -10358,6 +10463,7 @@ func (c FfiConverterReaderOptions) LowerExternal(value ReaderOptions) ExternalCR func (c FfiConverterReaderOptions) Write(writer io.Writer, value ReaderOptions) { FfiConverterUint64INSTANCE.Write(writer, value.ManifestPollIntervalMs) + FfiConverterUint64INSTANCE.Write(writer, value.WalPollIntervalMs) FfiConverterUint64INSTANCE.Write(writer, value.CheckpointLifetimeMs) FfiConverterUint64INSTANCE.Write(writer, value.MaxMemtableBytes) FfiConverterBoolINSTANCE.Write(writer, value.SkipWalReplay) @@ -10455,6 +10561,8 @@ type ScanOptions struct { // Optional context forwarded to custom filter policies; ignored by // built-in filters. Only consulted for prefix scans. FilterContext *FilterContext + // Optional caller-supplied tracing settings. + TracingOptions *TracingOptions } func (r *ScanOptions) Destroy() { @@ -10465,6 +10573,7 @@ func (r *ScanOptions) Destroy() { FfiDestroyerUint64{}.Destroy(r.MaxFetchTasks) FfiDestroyerOptionalIterationOrder{}.Destroy(r.Order) FfiDestroyerOptionalFilterContext{}.Destroy(r.FilterContext) + FfiDestroyerOptionalTracingOptions{}.Destroy(r.TracingOptions) } type FfiConverterScanOptions struct{} @@ -10484,6 +10593,7 @@ func (c FfiConverterScanOptions) Read(reader io.Reader) ScanOptions { FfiConverterUint64INSTANCE.Read(reader), FfiConverterOptionalIterationOrderINSTANCE.Read(reader), FfiConverterOptionalFilterContextINSTANCE.Read(reader), + FfiConverterOptionalTracingOptionsINSTANCE.Read(reader), } } @@ -10503,6 +10613,7 @@ func (c FfiConverterScanOptions) Write(writer io.Writer, value ScanOptions) { FfiConverterUint64INSTANCE.Write(writer, value.MaxFetchTasks) FfiConverterOptionalIterationOrderINSTANCE.Write(writer, value.Order) FfiConverterOptionalFilterContextINSTANCE.Write(writer, value.FilterContext) + FfiConverterOptionalTracingOptionsINSTANCE.Write(writer, value.TracingOptions) } type FfiDestroyerScanOptions struct{} @@ -10612,6 +10723,58 @@ func (_ FfiDestroyerSegmentPrefix) Destroy(value SegmentPrefix) { value.Destroy() } +// Options controlling how the native SlateDB WAL reader fetches WAL SSTs. +type SlateDbWalReaderOptions struct { + // Number of WAL SSTs to preload. + SstBatchSize uint64 + // Number of concurrent fetch tasks per WAL SST. + MaxFetchTasks uint64 + // Number of bytes to read ahead from each WAL SST. + ReadAheadBytes uint64 +} + +func (r *SlateDbWalReaderOptions) Destroy() { + FfiDestroyerUint64{}.Destroy(r.SstBatchSize) + FfiDestroyerUint64{}.Destroy(r.MaxFetchTasks) + FfiDestroyerUint64{}.Destroy(r.ReadAheadBytes) +} + +type FfiConverterSlateDbWalReaderOptions struct{} + +var FfiConverterSlateDbWalReaderOptionsINSTANCE = FfiConverterSlateDbWalReaderOptions{} + +func (c FfiConverterSlateDbWalReaderOptions) Lift(rb RustBufferI) SlateDbWalReaderOptions { + return LiftFromRustBuffer[SlateDbWalReaderOptions](c, rb) +} + +func (c FfiConverterSlateDbWalReaderOptions) Read(reader io.Reader) SlateDbWalReaderOptions { + return SlateDbWalReaderOptions{ + FfiConverterUint64INSTANCE.Read(reader), + FfiConverterUint64INSTANCE.Read(reader), + FfiConverterUint64INSTANCE.Read(reader), + } +} + +func (c FfiConverterSlateDbWalReaderOptions) Lower(value SlateDbWalReaderOptions) C.RustBuffer { + return LowerIntoRustBuffer[SlateDbWalReaderOptions](c, value) +} + +func (c FfiConverterSlateDbWalReaderOptions) LowerExternal(value SlateDbWalReaderOptions) ExternalCRustBuffer { + return RustBufferFromC(LowerIntoRustBuffer[SlateDbWalReaderOptions](c, value)) +} + +func (c FfiConverterSlateDbWalReaderOptions) Write(writer io.Writer, value SlateDbWalReaderOptions) { + FfiConverterUint64INSTANCE.Write(writer, value.SstBatchSize) + FfiConverterUint64INSTANCE.Write(writer, value.MaxFetchTasks) + FfiConverterUint64INSTANCE.Write(writer, value.ReadAheadBytes) +} + +type FfiDestroyerSlateDbWalReaderOptions struct{} + +func (_ FfiDestroyerSlateDbWalReaderOptions) Destroy(value SlateDbWalReaderOptions) { + value.Destroy() +} + // A sorted run made up of one or more SST views. type SortedRun struct { // Sorted run ID. @@ -10865,6 +11028,47 @@ func (_ FfiDestroyerSsTableView) Destroy(value SsTableView) { value.Destroy() } +// Options for tracing a read operation. +type TracingOptions struct { + TraceId string +} + +func (r *TracingOptions) Destroy() { + FfiDestroyerString{}.Destroy(r.TraceId) +} + +type FfiConverterTracingOptions struct{} + +var FfiConverterTracingOptionsINSTANCE = FfiConverterTracingOptions{} + +func (c FfiConverterTracingOptions) Lift(rb RustBufferI) TracingOptions { + return LiftFromRustBuffer[TracingOptions](c, rb) +} + +func (c FfiConverterTracingOptions) Read(reader io.Reader) TracingOptions { + return TracingOptions{ + FfiConverterStringINSTANCE.Read(reader), + } +} + +func (c FfiConverterTracingOptions) Lower(value TracingOptions) C.RustBuffer { + return LowerIntoRustBuffer[TracingOptions](c, value) +} + +func (c FfiConverterTracingOptions) LowerExternal(value TracingOptions) ExternalCRustBuffer { + return RustBufferFromC(LowerIntoRustBuffer[TracingOptions](c, value)) +} + +func (c FfiConverterTracingOptions) Write(writer io.Writer, value TracingOptions) { + FfiConverterStringINSTANCE.Write(writer, value.TraceId) +} + +type FfiDestroyerTracingOptions struct{} + +func (_ FfiDestroyerTracingOptions) Destroy(value TracingOptions) { + value.Destroy() +} + // A compactions snapshot paired with its version ID. type VersionedCompactions struct { // Compactions file version ID. @@ -11039,61 +11243,64 @@ func (_ FfiDestroyerVersionedManifest) Destroy(value VersionedManifest) { value.Destroy() } -// Metadata returned by a successful write. -type WriteHandle struct { - // Sequence number assigned to the write. - Seqnum uint64 - // Creation timestamp assigned to the write. - CreateTs int64 +// Rows from one fully consumed WAL file. +type WalRows struct { + // Rows stored in the WAL file. Empty fence WALs produce an empty vector. + Rows []RowEntry + // Last WAL file ID fully consumed by this batch. + LastConsumedWalFileId uint64 } -func (r *WriteHandle) Destroy() { - FfiDestroyerUint64{}.Destroy(r.Seqnum) - FfiDestroyerInt64{}.Destroy(r.CreateTs) +func (r *WalRows) Destroy() { + FfiDestroyerSequenceRowEntry{}.Destroy(r.Rows) + FfiDestroyerUint64{}.Destroy(r.LastConsumedWalFileId) } -type FfiConverterWriteHandle struct{} +type FfiConverterWalRows struct{} -var FfiConverterWriteHandleINSTANCE = FfiConverterWriteHandle{} +var FfiConverterWalRowsINSTANCE = FfiConverterWalRows{} -func (c FfiConverterWriteHandle) Lift(rb RustBufferI) WriteHandle { - return LiftFromRustBuffer[WriteHandle](c, rb) +func (c FfiConverterWalRows) Lift(rb RustBufferI) WalRows { + return LiftFromRustBuffer[WalRows](c, rb) } -func (c FfiConverterWriteHandle) Read(reader io.Reader) WriteHandle { - return WriteHandle{ +func (c FfiConverterWalRows) Read(reader io.Reader) WalRows { + return WalRows{ + FfiConverterSequenceRowEntryINSTANCE.Read(reader), FfiConverterUint64INSTANCE.Read(reader), - FfiConverterInt64INSTANCE.Read(reader), } } -func (c FfiConverterWriteHandle) Lower(value WriteHandle) C.RustBuffer { - return LowerIntoRustBuffer[WriteHandle](c, value) +func (c FfiConverterWalRows) Lower(value WalRows) C.RustBuffer { + return LowerIntoRustBuffer[WalRows](c, value) } -func (c FfiConverterWriteHandle) LowerExternal(value WriteHandle) ExternalCRustBuffer { - return RustBufferFromC(LowerIntoRustBuffer[WriteHandle](c, value)) +func (c FfiConverterWalRows) LowerExternal(value WalRows) ExternalCRustBuffer { + return RustBufferFromC(LowerIntoRustBuffer[WalRows](c, value)) } -func (c FfiConverterWriteHandle) Write(writer io.Writer, value WriteHandle) { - FfiConverterUint64INSTANCE.Write(writer, value.Seqnum) - FfiConverterInt64INSTANCE.Write(writer, value.CreateTs) +func (c FfiConverterWalRows) Write(writer io.Writer, value WalRows) { + FfiConverterSequenceRowEntryINSTANCE.Write(writer, value.Rows) + FfiConverterUint64INSTANCE.Write(writer, value.LastConsumedWalFileId) } -type FfiDestroyerWriteHandle struct{} +type FfiDestroyerWalRows struct{} -func (_ FfiDestroyerWriteHandle) Destroy(value WriteHandle) { +func (_ FfiDestroyerWalRows) Destroy(value WalRows) { value.Destroy() } -// Options that control durability behavior for writes and commits. +// Options that control writes and commits. type WriteOptions struct { // Whether the call waits for the write to become durable before returning. AwaitDurable bool + // Optional caller-supplied sequence number. Zero uses SlateDB's sequence oracle. + Seqnum uint64 } func (r *WriteOptions) Destroy() { FfiDestroyerBool{}.Destroy(r.AwaitDurable) + FfiDestroyerUint64{}.Destroy(r.Seqnum) } type FfiConverterWriteOptions struct{} @@ -11107,6 +11314,7 @@ func (c FfiConverterWriteOptions) Lift(rb RustBufferI) WriteOptions { func (c FfiConverterWriteOptions) Read(reader io.Reader) WriteOptions { return WriteOptions{ FfiConverterBoolINSTANCE.Read(reader), + FfiConverterUint64INSTANCE.Read(reader), } } @@ -11120,6 +11328,7 @@ func (c FfiConverterWriteOptions) LowerExternal(value WriteOptions) ExternalCRus func (c FfiConverterWriteOptions) Write(writer io.Writer, value WriteOptions) { FfiConverterBoolINSTANCE.Write(writer, value.AwaitDurable) + FfiConverterUint64INSTANCE.Write(writer, value.Seqnum) } type FfiDestroyerWriteOptions struct{} @@ -12714,21 +12923,21 @@ type TtlNoExpiry struct { func (e TtlNoExpiry) Destroy() { } -// Expire the value after the given number of clock ticks. -type TtlExpireAfterTicks struct { +// Expire the value after the given number of milliseconds. +type TtlExpireAfterMillis struct { Field0 uint64 } -func (e TtlExpireAfterTicks) Destroy() { +func (e TtlExpireAfterMillis) Destroy() { FfiDestroyerUint64{}.Destroy(e.Field0) } -// Expire the value at the given absolute timestamp (clock ticks). -type TtlExpireAt struct { +// Expire the value at the given Unix timestamp in milliseconds. +type TtlExpireAtMillis struct { Field0 int64 } -func (e TtlExpireAt) Destroy() { +func (e TtlExpireAtMillis) Destroy() { FfiDestroyerInt64{}.Destroy(e.Field0) } @@ -12755,11 +12964,11 @@ func (FfiConverterTtl) Read(reader io.Reader) Ttl { case 2: return TtlNoExpiry{} case 3: - return TtlExpireAfterTicks{ + return TtlExpireAfterMillis{ FfiConverterUint64INSTANCE.Read(reader), } case 4: - return TtlExpireAt{ + return TtlExpireAtMillis{ FfiConverterInt64INSTANCE.Read(reader), } default: @@ -12773,10 +12982,10 @@ func (FfiConverterTtl) Write(writer io.Writer, value Ttl) { writeInt32(writer, 1) case TtlNoExpiry: writeInt32(writer, 2) - case TtlExpireAfterTicks: + case TtlExpireAfterMillis: writeInt32(writer, 3) FfiConverterUint64INSTANCE.Write(writer, variant_value.Field0) - case TtlExpireAt: + case TtlExpireAtMillis: writeInt32(writer, 4) FfiConverterInt64INSTANCE.Write(writer, variant_value.Field0) default: @@ -13078,6 +13287,47 @@ func (_ FfiDestroyerOptionalPrefixExtractor) Destroy(value *PrefixExtractor) { } } +type FfiConverterOptionalWriteHandle struct{} + +var FfiConverterOptionalWriteHandleINSTANCE = FfiConverterOptionalWriteHandle{} + +func (c FfiConverterOptionalWriteHandle) Lift(rb RustBufferI) **WriteHandle { + return LiftFromRustBuffer[**WriteHandle](c, rb) +} + +func (_ FfiConverterOptionalWriteHandle) Read(reader io.Reader) **WriteHandle { + if readInt8(reader) == 0 { + return nil + } + temp := FfiConverterWriteHandleINSTANCE.Read(reader) + return &temp +} + +func (c FfiConverterOptionalWriteHandle) Lower(value **WriteHandle) C.RustBuffer { + return LowerIntoRustBuffer[**WriteHandle](c, value) +} + +func (c FfiConverterOptionalWriteHandle) LowerExternal(value **WriteHandle) ExternalCRustBuffer { + return RustBufferFromC(LowerIntoRustBuffer[**WriteHandle](c, value)) +} + +func (_ FfiConverterOptionalWriteHandle) Write(writer io.Writer, value **WriteHandle) { + if value == nil { + writeInt8(writer, 0) + } else { + writeInt8(writer, 1) + FfiConverterWriteHandleINSTANCE.Write(writer, *value) + } +} + +type FfiDestroyerOptionalWriteHandle struct{} + +func (_ FfiDestroyerOptionalWriteHandle) Destroy(value **WriteHandle) { + if value != nil { + FfiDestroyerWriteHandle{}.Destroy(*value) + } +} + type FfiConverterOptionalCompaction struct{} var FfiConverterOptionalCompactionINSTANCE = FfiConverterOptionalCompaction{} @@ -13365,44 +13615,44 @@ func (_ FfiDestroyerOptionalMetric) Destroy(value *Metric) { } } -type FfiConverterOptionalRowEntry struct{} +type FfiConverterOptionalTracingOptions struct{} -var FfiConverterOptionalRowEntryINSTANCE = FfiConverterOptionalRowEntry{} +var FfiConverterOptionalTracingOptionsINSTANCE = FfiConverterOptionalTracingOptions{} -func (c FfiConverterOptionalRowEntry) Lift(rb RustBufferI) *RowEntry { - return LiftFromRustBuffer[*RowEntry](c, rb) +func (c FfiConverterOptionalTracingOptions) Lift(rb RustBufferI) *TracingOptions { + return LiftFromRustBuffer[*TracingOptions](c, rb) } -func (_ FfiConverterOptionalRowEntry) Read(reader io.Reader) *RowEntry { +func (_ FfiConverterOptionalTracingOptions) Read(reader io.Reader) *TracingOptions { if readInt8(reader) == 0 { return nil } - temp := FfiConverterRowEntryINSTANCE.Read(reader) + temp := FfiConverterTracingOptionsINSTANCE.Read(reader) return &temp } -func (c FfiConverterOptionalRowEntry) Lower(value *RowEntry) C.RustBuffer { - return LowerIntoRustBuffer[*RowEntry](c, value) +func (c FfiConverterOptionalTracingOptions) Lower(value *TracingOptions) C.RustBuffer { + return LowerIntoRustBuffer[*TracingOptions](c, value) } -func (c FfiConverterOptionalRowEntry) LowerExternal(value *RowEntry) ExternalCRustBuffer { - return RustBufferFromC(LowerIntoRustBuffer[*RowEntry](c, value)) +func (c FfiConverterOptionalTracingOptions) LowerExternal(value *TracingOptions) ExternalCRustBuffer { + return RustBufferFromC(LowerIntoRustBuffer[*TracingOptions](c, value)) } -func (_ FfiConverterOptionalRowEntry) Write(writer io.Writer, value *RowEntry) { +func (_ FfiConverterOptionalTracingOptions) Write(writer io.Writer, value *TracingOptions) { if value == nil { writeInt8(writer, 0) } else { writeInt8(writer, 1) - FfiConverterRowEntryINSTANCE.Write(writer, *value) + FfiConverterTracingOptionsINSTANCE.Write(writer, *value) } } -type FfiDestroyerOptionalRowEntry struct{} +type FfiDestroyerOptionalTracingOptions struct{} -func (_ FfiDestroyerOptionalRowEntry) Destroy(value *RowEntry) { +func (_ FfiDestroyerOptionalTracingOptions) Destroy(value *TracingOptions) { if value != nil { - FfiDestroyerRowEntry{}.Destroy(*value) + FfiDestroyerTracingOptions{}.Destroy(*value) } } @@ -13488,44 +13738,44 @@ func (_ FfiDestroyerOptionalVersionedManifest) Destroy(value *VersionedManifest) } } -type FfiConverterOptionalWriteHandle struct{} +type FfiConverterOptionalWalRows struct{} -var FfiConverterOptionalWriteHandleINSTANCE = FfiConverterOptionalWriteHandle{} +var FfiConverterOptionalWalRowsINSTANCE = FfiConverterOptionalWalRows{} -func (c FfiConverterOptionalWriteHandle) Lift(rb RustBufferI) *WriteHandle { - return LiftFromRustBuffer[*WriteHandle](c, rb) +func (c FfiConverterOptionalWalRows) Lift(rb RustBufferI) *WalRows { + return LiftFromRustBuffer[*WalRows](c, rb) } -func (_ FfiConverterOptionalWriteHandle) Read(reader io.Reader) *WriteHandle { +func (_ FfiConverterOptionalWalRows) Read(reader io.Reader) *WalRows { if readInt8(reader) == 0 { return nil } - temp := FfiConverterWriteHandleINSTANCE.Read(reader) + temp := FfiConverterWalRowsINSTANCE.Read(reader) return &temp } -func (c FfiConverterOptionalWriteHandle) Lower(value *WriteHandle) C.RustBuffer { - return LowerIntoRustBuffer[*WriteHandle](c, value) +func (c FfiConverterOptionalWalRows) Lower(value *WalRows) C.RustBuffer { + return LowerIntoRustBuffer[*WalRows](c, value) } -func (c FfiConverterOptionalWriteHandle) LowerExternal(value *WriteHandle) ExternalCRustBuffer { - return RustBufferFromC(LowerIntoRustBuffer[*WriteHandle](c, value)) +func (c FfiConverterOptionalWalRows) LowerExternal(value *WalRows) ExternalCRustBuffer { + return RustBufferFromC(LowerIntoRustBuffer[*WalRows](c, value)) } -func (_ FfiConverterOptionalWriteHandle) Write(writer io.Writer, value *WriteHandle) { +func (_ FfiConverterOptionalWalRows) Write(writer io.Writer, value *WalRows) { if value == nil { writeInt8(writer, 0) } else { writeInt8(writer, 1) - FfiConverterWriteHandleINSTANCE.Write(writer, *value) + FfiConverterWalRowsINSTANCE.Write(writer, *value) } } -type FfiDestroyerOptionalWriteHandle struct{} +type FfiDestroyerOptionalWalRows struct{} -func (_ FfiDestroyerOptionalWriteHandle) Destroy(value *WriteHandle) { +func (_ FfiDestroyerOptionalWalRows) Destroy(value *WalRows) { if value != nil { - FfiDestroyerWriteHandle{}.Destroy(*value) + FfiDestroyerWalRows{}.Destroy(*value) } } @@ -13652,6 +13902,47 @@ func (_ FfiDestroyerOptionalFilterContext) Destroy(value *FilterContext) { } } +type FfiConverterOptionalFlushType struct{} + +var FfiConverterOptionalFlushTypeINSTANCE = FfiConverterOptionalFlushType{} + +func (c FfiConverterOptionalFlushType) Lift(rb RustBufferI) *FlushType { + return LiftFromRustBuffer[*FlushType](c, rb) +} + +func (_ FfiConverterOptionalFlushType) Read(reader io.Reader) *FlushType { + if readInt8(reader) == 0 { + return nil + } + temp := FfiConverterFlushTypeINSTANCE.Read(reader) + return &temp +} + +func (c FfiConverterOptionalFlushType) Lower(value *FlushType) C.RustBuffer { + return LowerIntoRustBuffer[*FlushType](c, value) +} + +func (c FfiConverterOptionalFlushType) LowerExternal(value *FlushType) ExternalCRustBuffer { + return RustBufferFromC(LowerIntoRustBuffer[*FlushType](c, value)) +} + +func (_ FfiConverterOptionalFlushType) Write(writer io.Writer, value *FlushType) { + if value == nil { + writeInt8(writer, 0) + } else { + writeInt8(writer, 1) + FfiConverterFlushTypeINSTANCE.Write(writer, *value) + } +} + +type FfiDestroyerOptionalFlushType struct{} + +func (_ FfiDestroyerOptionalFlushType) Destroy(value *FlushType) { + if value != nil { + FfiDestroyerFlushType{}.Destroy(*value) + } +} + type FfiConverterOptionalIterationOrder struct{} var FfiConverterOptionalIterationOrderINSTANCE = FfiConverterOptionalIterationOrder{} @@ -13881,53 +14172,6 @@ func (FfiDestroyerSequenceFilterPolicy) Destroy(sequence []*FilterPolicy) { } } -type FfiConverterSequenceWalFile struct{} - -var FfiConverterSequenceWalFileINSTANCE = FfiConverterSequenceWalFile{} - -func (c FfiConverterSequenceWalFile) Lift(rb RustBufferI) []*WalFile { - return LiftFromRustBuffer[[]*WalFile](c, rb) -} - -func (c FfiConverterSequenceWalFile) Read(reader io.Reader) []*WalFile { - length := readInt32(reader) - if length == 0 { - return nil - } - result := make([]*WalFile, 0, length) - for i := int32(0); i < length; i++ { - result = append(result, FfiConverterWalFileINSTANCE.Read(reader)) - } - return result -} - -func (c FfiConverterSequenceWalFile) Lower(value []*WalFile) C.RustBuffer { - return LowerIntoRustBuffer[[]*WalFile](c, value) -} - -func (c FfiConverterSequenceWalFile) LowerExternal(value []*WalFile) ExternalCRustBuffer { - return RustBufferFromC(LowerIntoRustBuffer[[]*WalFile](c, value)) -} - -func (c FfiConverterSequenceWalFile) Write(writer io.Writer, value []*WalFile) { - if len(value) > math.MaxInt32 { - panic("[]*WalFile is too large to fit into Int32") - } - - writeInt32(writer, int32(len(value))) - for _, item := range value { - FfiConverterWalFileINSTANCE.Write(writer, item) - } -} - -type FfiDestroyerSequenceWalFile struct{} - -func (FfiDestroyerSequenceWalFile) Destroy(sequence []*WalFile) { - for _, value := range sequence { - FfiDestroyerWalFile{}.Destroy(value) - } -} - type FfiConverterSequenceCheckpoint struct{} var FfiConverterSequenceCheckpointINSTANCE = FfiConverterSequenceCheckpoint{} @@ -14210,6 +14454,53 @@ func (FfiDestroyerSequenceMetricLabel) Destroy(sequence []MetricLabel) { } } +type FfiConverterSequenceRowEntry struct{} + +var FfiConverterSequenceRowEntryINSTANCE = FfiConverterSequenceRowEntry{} + +func (c FfiConverterSequenceRowEntry) Lift(rb RustBufferI) []RowEntry { + return LiftFromRustBuffer[[]RowEntry](c, rb) +} + +func (c FfiConverterSequenceRowEntry) Read(reader io.Reader) []RowEntry { + length := readInt32(reader) + if length == 0 { + return nil + } + result := make([]RowEntry, 0, length) + for i := int32(0); i < length; i++ { + result = append(result, FfiConverterRowEntryINSTANCE.Read(reader)) + } + return result +} + +func (c FfiConverterSequenceRowEntry) Lower(value []RowEntry) C.RustBuffer { + return LowerIntoRustBuffer[[]RowEntry](c, value) +} + +func (c FfiConverterSequenceRowEntry) LowerExternal(value []RowEntry) ExternalCRustBuffer { + return RustBufferFromC(LowerIntoRustBuffer[[]RowEntry](c, value)) +} + +func (c FfiConverterSequenceRowEntry) Write(writer io.Writer, value []RowEntry) { + if len(value) > math.MaxInt32 { + panic("[]RowEntry is too large to fit into Int32") + } + + writeInt32(writer, int32(len(value))) + for _, item := range value { + FfiConverterRowEntryINSTANCE.Write(writer, item) + } +} + +type FfiDestroyerSequenceRowEntry struct{} + +func (FfiDestroyerSequenceRowEntry) Destroy(sequence []RowEntry) { + for _, value := range sequence { + FfiDestroyerRowEntry{}.Destroy(value) + } +} + type FfiConverterSequenceSegment struct{} var FfiConverterSequenceSegmentINSTANCE = FfiConverterSequenceSegment{} diff --git a/bindings/go/uniffi/slatedb.h b/bindings/go/uniffi/slatedb.h index 5d21f81ceb..8ad45f4306 100644 --- a/bindings/go/uniffi/slatedb.h +++ b/bindings/go/uniffi/slatedb.h @@ -1004,6 +1004,11 @@ uint64_t uniffi_slatedb_uniffi_fn_method_db_scan_with_options(uint64_t ptr, Rust uint64_t uniffi_slatedb_uniffi_fn_method_db_shutdown(uint64_t ptr ); #endif +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_DB_SHUTDOWN_WITH_OPTIONS +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_DB_SHUTDOWN_WITH_OPTIONS +uint64_t uniffi_slatedb_uniffi_fn_method_db_shutdown_with_options(uint64_t ptr, RustBuffer options +); +#endif #ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_DB_SNAPSHOT #define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_DB_SNAPSHOT uint64_t uniffi_slatedb_uniffi_fn_method_db_snapshot(uint64_t ptr @@ -1637,79 +1642,59 @@ void uniffi_slatedb_uniffi_fn_method_settings_set(uint64_t ptr, RustBuffer key, RustBuffer uniffi_slatedb_uniffi_fn_method_settings_to_json_string(uint64_t ptr, RustCallStatus *out_status ); #endif -#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_CLONE_WALFILE -#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_CLONE_WALFILE -uint64_t uniffi_slatedb_uniffi_fn_clone_walfile(uint64_t handle, RustCallStatus *out_status -); -#endif -#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_FREE_WALFILE -#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_FREE_WALFILE -void uniffi_slatedb_uniffi_fn_free_walfile(uint64_t handle, RustCallStatus *out_status -); -#endif -#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_WALFILE_ID -#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_WALFILE_ID -uint64_t uniffi_slatedb_uniffi_fn_method_walfile_id(uint64_t ptr, RustCallStatus *out_status -); -#endif -#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_WALFILE_ITERATOR -#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_WALFILE_ITERATOR -uint64_t uniffi_slatedb_uniffi_fn_method_walfile_iterator(uint64_t ptr -); -#endif -#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_WALFILE_METADATA -#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_WALFILE_METADATA -uint64_t uniffi_slatedb_uniffi_fn_method_walfile_metadata(uint64_t ptr +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_CLONE_SLATEDBWALITERATOR +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_CLONE_SLATEDBWALITERATOR +uint64_t uniffi_slatedb_uniffi_fn_clone_slatedbwaliterator(uint64_t handle, RustCallStatus *out_status ); #endif -#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_WALFILE_NEXT_FILE -#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_WALFILE_NEXT_FILE -uint64_t uniffi_slatedb_uniffi_fn_method_walfile_next_file(uint64_t ptr, RustCallStatus *out_status +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_FREE_SLATEDBWALITERATOR +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_FREE_SLATEDBWALITERATOR +void uniffi_slatedb_uniffi_fn_free_slatedbwaliterator(uint64_t handle, RustCallStatus *out_status ); #endif -#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_WALFILE_NEXT_ID -#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_WALFILE_NEXT_ID -uint64_t uniffi_slatedb_uniffi_fn_method_walfile_next_id(uint64_t ptr, RustCallStatus *out_status +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_SLATEDBWALITERATOR_NEXT +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_SLATEDBWALITERATOR_NEXT +uint64_t uniffi_slatedb_uniffi_fn_method_slatedbwaliterator_next(uint64_t ptr ); #endif -#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_CLONE_WALFILEITERATOR -#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_CLONE_WALFILEITERATOR -uint64_t uniffi_slatedb_uniffi_fn_clone_walfileiterator(uint64_t handle, RustCallStatus *out_status +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_CLONE_SLATEDBWALREADER +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_CLONE_SLATEDBWALREADER +uint64_t uniffi_slatedb_uniffi_fn_clone_slatedbwalreader(uint64_t handle, RustCallStatus *out_status ); #endif -#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_FREE_WALFILEITERATOR -#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_FREE_WALFILEITERATOR -void uniffi_slatedb_uniffi_fn_free_walfileiterator(uint64_t handle, RustCallStatus *out_status +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_FREE_SLATEDBWALREADER +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_FREE_SLATEDBWALREADER +void uniffi_slatedb_uniffi_fn_free_slatedbwalreader(uint64_t handle, RustCallStatus *out_status ); #endif -#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_WALFILEITERATOR_NEXT -#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_WALFILEITERATOR_NEXT -uint64_t uniffi_slatedb_uniffi_fn_method_walfileiterator_next(uint64_t ptr +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_CONSTRUCTOR_SLATEDBWALREADER_NEW +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_CONSTRUCTOR_SLATEDBWALREADER_NEW +uint64_t uniffi_slatedb_uniffi_fn_constructor_slatedbwalreader_new(RustBuffer path, uint64_t object_store, RustCallStatus *out_status ); #endif -#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_CLONE_WALREADER -#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_CLONE_WALREADER -uint64_t uniffi_slatedb_uniffi_fn_clone_walreader(uint64_t handle, RustCallStatus *out_status +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_CONSTRUCTOR_SLATEDBWALREADER_WITH_OPTIONS +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_CONSTRUCTOR_SLATEDBWALREADER_WITH_OPTIONS +uint64_t uniffi_slatedb_uniffi_fn_constructor_slatedbwalreader_with_options(RustBuffer path, uint64_t object_store, RustBuffer options, RustCallStatus *out_status ); #endif -#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_FREE_WALREADER -#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_FREE_WALREADER -void uniffi_slatedb_uniffi_fn_free_walreader(uint64_t handle, RustCallStatus *out_status +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_CONSTRUCTOR_SLATEDBWALREADER_WITH_WAL_OBJECT_STORE +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_CONSTRUCTOR_SLATEDBWALREADER_WITH_WAL_OBJECT_STORE +uint64_t uniffi_slatedb_uniffi_fn_constructor_slatedbwalreader_with_wal_object_store(RustBuffer path, uint64_t object_store, uint64_t wal_object_store, RustCallStatus *out_status ); #endif -#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_CONSTRUCTOR_WALREADER_NEW -#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_CONSTRUCTOR_WALREADER_NEW -uint64_t uniffi_slatedb_uniffi_fn_constructor_walreader_new(RustBuffer path, uint64_t object_store, RustCallStatus *out_status +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_CONSTRUCTOR_SLATEDBWALREADER_WITH_WAL_OBJECT_STORE_AND_OPTIONS +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_CONSTRUCTOR_SLATEDBWALREADER_WITH_WAL_OBJECT_STORE_AND_OPTIONS +uint64_t uniffi_slatedb_uniffi_fn_constructor_slatedbwalreader_with_wal_object_store_and_options(RustBuffer path, uint64_t object_store, uint64_t wal_object_store, RustBuffer options, RustCallStatus *out_status ); #endif -#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_WALREADER_GET -#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_WALREADER_GET -uint64_t uniffi_slatedb_uniffi_fn_method_walreader_get(uint64_t ptr, uint64_t id, RustCallStatus *out_status +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_SLATEDBWALREADER_ITERATOR +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_SLATEDBWALREADER_ITERATOR +uint64_t uniffi_slatedb_uniffi_fn_method_slatedbwalreader_iterator(uint64_t ptr, uint64_t start_wal_file_id ); #endif -#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_WALREADER_LIST -#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_WALREADER_LIST -uint64_t uniffi_slatedb_uniffi_fn_method_walreader_list(uint64_t ptr, RustBuffer start_id, RustBuffer end_id +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_SLATEDBWALREADER_LAST_WAL_FILE_ID +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_SLATEDBWALREADER_LAST_WAL_FILE_ID +uint64_t uniffi_slatedb_uniffi_fn_method_slatedbwalreader_last_wal_file_id(uint64_t ptr, uint64_t replay_after_wal_id ); #endif #ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_CLONE_WRITEBATCH @@ -1753,6 +1738,31 @@ void uniffi_slatedb_uniffi_fn_method_writebatch_put(uint64_t ptr, RustBuffer key void uniffi_slatedb_uniffi_fn_method_writebatch_put_with_options(uint64_t ptr, RustBuffer key, RustBuffer value, RustBuffer options, RustCallStatus *out_status ); #endif +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_CLONE_WRITEHANDLE +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_CLONE_WRITEHANDLE +uint64_t uniffi_slatedb_uniffi_fn_clone_writehandle(uint64_t handle, RustCallStatus *out_status +); +#endif +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_FREE_WRITEHANDLE +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_FREE_WRITEHANDLE +void uniffi_slatedb_uniffi_fn_free_writehandle(uint64_t handle, RustCallStatus *out_status +); +#endif +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_WRITEHANDLE_AWAIT_DURABLE +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_WRITEHANDLE_AWAIT_DURABLE +uint64_t uniffi_slatedb_uniffi_fn_method_writehandle_await_durable(uint64_t ptr +); +#endif +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_WRITEHANDLE_CREATE_TS +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_WRITEHANDLE_CREATE_TS +int64_t uniffi_slatedb_uniffi_fn_method_writehandle_create_ts(uint64_t ptr, RustCallStatus *out_status +); +#endif +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_WRITEHANDLE_SEQNUM +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_METHOD_WRITEHANDLE_SEQNUM +uint64_t uniffi_slatedb_uniffi_fn_method_writehandle_seqnum(uint64_t ptr, RustCallStatus *out_status +); +#endif #ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_FUNC_INIT_LOGGING #define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_FN_FUNC_INIT_LOGGING void uniffi_slatedb_uniffi_fn_func_init_logging(RustBuffer level, RustBuffer callback, RustCallStatus *out_status @@ -2400,6 +2410,12 @@ uint16_t uniffi_slatedb_uniffi_checksum_method_db_scan_with_options(void #define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_DB_SHUTDOWN uint16_t uniffi_slatedb_uniffi_checksum_method_db_shutdown(void +); +#endif +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_DB_SHUTDOWN_WITH_OPTIONS +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_DB_SHUTDOWN_WITH_OPTIONS +uint16_t uniffi_slatedb_uniffi_checksum_method_db_shutdown_with_options(void + ); #endif #ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_DB_SNAPSHOT @@ -2822,51 +2838,21 @@ uint16_t uniffi_slatedb_uniffi_checksum_method_settings_to_json_string(void ); #endif -#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_WALFILE_ID -#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_WALFILE_ID -uint16_t uniffi_slatedb_uniffi_checksum_method_walfile_id(void - -); -#endif -#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_WALFILE_ITERATOR -#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_WALFILE_ITERATOR -uint16_t uniffi_slatedb_uniffi_checksum_method_walfile_iterator(void - -); -#endif -#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_WALFILE_METADATA -#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_WALFILE_METADATA -uint16_t uniffi_slatedb_uniffi_checksum_method_walfile_metadata(void - -); -#endif -#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_WALFILE_NEXT_FILE -#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_WALFILE_NEXT_FILE -uint16_t uniffi_slatedb_uniffi_checksum_method_walfile_next_file(void - -); -#endif -#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_WALFILE_NEXT_ID -#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_WALFILE_NEXT_ID -uint16_t uniffi_slatedb_uniffi_checksum_method_walfile_next_id(void +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_SLATEDBWALITERATOR_NEXT +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_SLATEDBWALITERATOR_NEXT +uint16_t uniffi_slatedb_uniffi_checksum_method_slatedbwaliterator_next(void ); #endif -#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_WALFILEITERATOR_NEXT -#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_WALFILEITERATOR_NEXT -uint16_t uniffi_slatedb_uniffi_checksum_method_walfileiterator_next(void +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_SLATEDBWALREADER_ITERATOR +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_SLATEDBWALREADER_ITERATOR +uint16_t uniffi_slatedb_uniffi_checksum_method_slatedbwalreader_iterator(void ); #endif -#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_WALREADER_GET -#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_WALREADER_GET -uint16_t uniffi_slatedb_uniffi_checksum_method_walreader_get(void - -); -#endif -#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_WALREADER_LIST -#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_WALREADER_LIST -uint16_t uniffi_slatedb_uniffi_checksum_method_walreader_list(void +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_SLATEDBWALREADER_LAST_WAL_FILE_ID +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_SLATEDBWALREADER_LAST_WAL_FILE_ID +uint16_t uniffi_slatedb_uniffi_checksum_method_slatedbwalreader_last_wal_file_id(void ); #endif @@ -2898,6 +2884,24 @@ uint16_t uniffi_slatedb_uniffi_checksum_method_writebatch_put(void #define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_WRITEBATCH_PUT_WITH_OPTIONS uint16_t uniffi_slatedb_uniffi_checksum_method_writebatch_put_with_options(void +); +#endif +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_WRITEHANDLE_AWAIT_DURABLE +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_WRITEHANDLE_AWAIT_DURABLE +uint16_t uniffi_slatedb_uniffi_checksum_method_writehandle_await_durable(void + +); +#endif +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_WRITEHANDLE_CREATE_TS +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_WRITEHANDLE_CREATE_TS +uint16_t uniffi_slatedb_uniffi_checksum_method_writehandle_create_ts(void + +); +#endif +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_WRITEHANDLE_SEQNUM +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_METHOD_WRITEHANDLE_SEQNUM +uint16_t uniffi_slatedb_uniffi_checksum_method_writehandle_seqnum(void + ); #endif #ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_CONSTRUCTOR_ADMINBUILDER_NEW @@ -3002,9 +3006,27 @@ uint16_t uniffi_slatedb_uniffi_checksum_constructor_settings_load(void ); #endif -#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_CONSTRUCTOR_WALREADER_NEW -#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_CONSTRUCTOR_WALREADER_NEW -uint16_t uniffi_slatedb_uniffi_checksum_constructor_walreader_new(void +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_CONSTRUCTOR_SLATEDBWALREADER_NEW +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_CONSTRUCTOR_SLATEDBWALREADER_NEW +uint16_t uniffi_slatedb_uniffi_checksum_constructor_slatedbwalreader_new(void + +); +#endif +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_CONSTRUCTOR_SLATEDBWALREADER_WITH_OPTIONS +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_CONSTRUCTOR_SLATEDBWALREADER_WITH_OPTIONS +uint16_t uniffi_slatedb_uniffi_checksum_constructor_slatedbwalreader_with_options(void + +); +#endif +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_CONSTRUCTOR_SLATEDBWALREADER_WITH_WAL_OBJECT_STORE +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_CONSTRUCTOR_SLATEDBWALREADER_WITH_WAL_OBJECT_STORE +uint16_t uniffi_slatedb_uniffi_checksum_constructor_slatedbwalreader_with_wal_object_store(void + +); +#endif +#ifndef UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_CONSTRUCTOR_SLATEDBWALREADER_WITH_WAL_OBJECT_STORE_AND_OPTIONS +#define UNIFFI_FFIDEF_UNIFFI_SLATEDB_UNIFFI_CHECKSUM_CONSTRUCTOR_SLATEDBWALREADER_WITH_WAL_OBJECT_STORE_AND_OPTIONS +uint16_t uniffi_slatedb_uniffi_checksum_constructor_slatedbwalreader_with_wal_object_store_and_options(void ); #endif diff --git a/bindings/go/uniffi/slatedb_test.go b/bindings/go/uniffi/slatedb_test.go index dfcc42dbed..95d199bcf6 100644 --- a/bindings/go/uniffi/slatedb_test.go +++ b/bindings/go/uniffi/slatedb_test.go @@ -137,12 +137,15 @@ func openTestReader(t *testing.T, store *slatedb.ObjectStore, configure func(*te return handle } -func openTestWalReader(t *testing.T, store *slatedb.ObjectStore) *slatedb.WalReader { +func openTestSlateDbWalReader(t *testing.T, store *slatedb.ObjectStore) *slatedb.SlateDbWalReader { t.Helper() - reader := slatedb.NewWalReader(testDBPath, store) + reader, err := slatedb.NewSlateDbWalReader(testDBPath, store) + if err != nil { + t.Fatalf("NewSlateDbWalReader(): %v", err) + } if reader == nil { - t.Fatal("NewWalReader(): got nil reader") + t.Fatal("NewSlateDbWalReader(): got nil reader") } t.Cleanup(reader.Destroy) @@ -158,6 +161,22 @@ func uint64Ptr(v uint64) *uint64 { return &v } +func trackWriteHandle(t *testing.T, handle *slatedb.WriteHandle) *slatedb.WriteHandle { + t.Helper() + if handle == nil { + t.Fatal("got nil write handle") + } + t.Cleanup(handle.Destroy) + return handle +} + +func awaitDurable(t *testing.T, handle *slatedb.WriteHandle) { + t.Helper() + if err := handle.AwaitDurable(); err != nil { + t.Fatalf("WriteHandle.AwaitDurable(): %v", err) + } +} + func drainIterator(t *testing.T, iter *slatedb.DbIterator) []slatedb.KeyValue { t.Helper() @@ -174,20 +193,25 @@ func drainIterator(t *testing.T, iter *slatedb.DbIterator) []slatedb.KeyValue { } } -func drainWalIterator(t *testing.T, iter *slatedb.WalFileIterator) []slatedb.RowEntry { +func readWalBatchesThrough( + t *testing.T, + iter *slatedb.SlateDbWalIterator, + endWalFileID uint64, +) []slatedb.WalRows { t.Helper() - var rows []slatedb.RowEntry - for { - row, err := iter.Next() + var batches []slatedb.WalRows + for len(batches) == 0 || batches[len(batches)-1].LastConsumedWalFileId < endWalFileID { + batch, err := iter.Next() if err != nil { t.Fatalf("wal iterator Next(): %v", err) } - if row == nil { - return rows + if batch == nil { + t.Fatal("live WAL iterator ended unexpectedly") } - rows = append(rows, *row) + batches = append(batches, *batch) } + return batches } func requireRows(t *testing.T, got []slatedb.KeyValue, wantKeys []string, wantValues []string) { @@ -451,6 +475,32 @@ func seedWalFiles(t *testing.T, store *slatedb.ObjectStore) { if err := handle.db.FlushWithOptions(slatedb.FlushOptions{FlushType: slatedb.FlushTypeWal}); err != nil { t.Fatalf("FlushWithOptions(Wal) for merge row: %v", err) } + if err := handle.db.Shutdown(); err != nil { + t.Fatalf("Shutdown() after seeding WAL files: %v", err) + } + handle.open = false +} + +func appendWalValue(t *testing.T, store *slatedb.ObjectStore, key, value string) { + t.Helper() + + handle := openTestDB(t, store, func(t *testing.T, builder *slatedb.DbBuilder) { + t.Helper() + if err := builder.WithMergeOperator(concatMergeOperator{}); err != nil { + t.Fatalf("WithMergeOperator(): %v", err) + } + }) + + if _, err := handle.db.Put([]byte(key), []byte(value)); err != nil { + t.Fatalf("Put(%s): %v", key, err) + } + if err := handle.db.FlushWithOptions(slatedb.FlushOptions{FlushType: slatedb.FlushTypeWal}); err != nil { + t.Fatalf("FlushWithOptions(Wal) for %s: %v", key, err) + } + if err := handle.db.Shutdown(); err != nil { + t.Fatalf("Shutdown() after appending %s: %v", key, err) + } + handle.open = false } func TestDbLifecycleAndStatus(t *testing.T) { @@ -504,6 +554,38 @@ func TestDbLifecycleAndStatus(t *testing.T) { } } +func TestDbShutdownWithOptions(t *testing.T) { + wal := slatedb.FlushTypeWal + tests := []struct { + name string + flushType *slatedb.FlushType + }{ + {name: "wal", flushType: &wal}, + {name: "none", flushType: nil}, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + store := newMemoryStore(t) + handle := openTestDB(t, store, nil) + + if _, err := handle.db.Put([]byte("shutdown-options"), []byte("value")); err != nil { + t.Fatalf("Put(): %v", err) + } + + if err := handle.db.ShutdownWithOptions(slatedb.CloseOptions{FlushType: tt.flushType}); err != nil { + t.Fatalf("ShutdownWithOptions(): %v", err) + } + handle.open = false + + status := handle.db.Status() + if status.CloseReason == nil || *status.CloseReason != slatedb.CloseReasonClean { + t.Fatalf("Status() after ShutdownWithOptions(): got close reason %v, want %v", status.CloseReason, slatedb.CloseReasonClean) + } + }) + } +} + type fixedThreeByteSegmentExtractor struct{} func (fixedThreeByteSegmentExtractor) Name() string { return "fixed_three_byte" } @@ -597,16 +679,17 @@ func TestDbCrudAndMetadata(t *testing.T) { } putOptions := slatedb.PutOptions{Ttl: slatedb.TtlDefault{}} - writeOptions := slatedb.WriteOptions{AwaitDurable: true} + writeOptions := slatedb.WriteOptions{AwaitDurable: true, Seqnum: 0} firstWrite, err := handle.db.Put([]byte("alpha"), []byte("one")) if err != nil { t.Fatalf("Put(alpha): %v", err) } - if firstWrite.Seqnum == 0 { + firstWrite = trackWriteHandle(t, firstWrite) + if firstWrite.Seqnum() == 0 { t.Fatalf("Put(alpha): Seqnum = 0") } - if firstWrite.CreateTs == 0 { + if firstWrite.CreateTs() == 0 { t.Fatalf("Put(alpha): CreateTs = 0") } @@ -639,11 +722,11 @@ func TestDbCrudAndMetadata(t *testing.T) { if !bytes.Equal(metadata.Value, []byte("one")) { t.Fatalf("GetKeyValue(alpha): value = %q, want %q", metadata.Value, "one") } - if metadata.Seq != firstWrite.Seqnum { - t.Fatalf("GetKeyValue(alpha): seq = %d, want %d", metadata.Seq, firstWrite.Seqnum) + if metadata.Seq != firstWrite.Seqnum() { + t.Fatalf("GetKeyValue(alpha): seq = %d, want %d", metadata.Seq, firstWrite.Seqnum()) } - if metadata.CreateTs != firstWrite.CreateTs { - t.Fatalf("GetKeyValue(alpha): create ts = %d, want %d", metadata.CreateTs, firstWrite.CreateTs) + if metadata.CreateTs != firstWrite.CreateTs() { + t.Fatalf("GetKeyValue(alpha): create ts = %d, want %d", metadata.CreateTs, firstWrite.CreateTs()) } metadata, err = handle.db.GetKeyValueWithOptions([]byte("alpha"), readOptions) @@ -658,10 +741,12 @@ func TestDbCrudAndMetadata(t *testing.T) { if err != nil { t.Fatalf("PutWithOptions(beta): %v", err) } - if secondWrite.Seqnum <= firstWrite.Seqnum { - t.Fatalf("PutWithOptions(beta): seq = %d, want > %d", secondWrite.Seqnum, firstWrite.Seqnum) + secondWrite = trackWriteHandle(t, secondWrite) + awaitDurable(t, secondWrite) + if secondWrite.Seqnum() <= firstWrite.Seqnum() { + t.Fatalf("PutWithOptions(beta): seq = %d, want > %d", secondWrite.Seqnum(), firstWrite.Seqnum()) } - if secondWrite.CreateTs == 0 { + if secondWrite.CreateTs() == 0 { t.Fatalf("PutWithOptions(beta): CreateTs = 0") } @@ -696,8 +781,9 @@ func TestDbCrudAndMetadata(t *testing.T) { if err != nil { t.Fatalf("Delete(alpha): %v", err) } - if deleteWrite.Seqnum <= secondWrite.Seqnum { - t.Fatalf("Delete(alpha): seq = %d, want > %d", deleteWrite.Seqnum, secondWrite.Seqnum) + deleteWrite = trackWriteHandle(t, deleteWrite) + if deleteWrite.Seqnum() <= secondWrite.Seqnum() { + t.Fatalf("Delete(alpha): seq = %d, want > %d", deleteWrite.Seqnum(), secondWrite.Seqnum()) } value, err = handle.db.Get([]byte("alpha")) @@ -712,8 +798,9 @@ func TestDbCrudAndMetadata(t *testing.T) { if err != nil { t.Fatalf("DeleteWithOptions(beta): %v", err) } - if deleteWrite.Seqnum <= secondWrite.Seqnum { - t.Fatalf("DeleteWithOptions(beta): seq = %d, want > %d", deleteWrite.Seqnum, secondWrite.Seqnum) + deleteWrite = trackWriteHandle(t, deleteWrite) + if deleteWrite.Seqnum() <= secondWrite.Seqnum() { + t.Fatalf("DeleteWithOptions(beta): seq = %d, want > %d", deleteWrite.Seqnum(), secondWrite.Seqnum()) } value, err = handle.db.Get([]byte("beta")) @@ -846,7 +933,8 @@ func TestDbBatchWriteAndConsumption(t *testing.T) { if err != nil { t.Fatalf("Write(): %v", err) } - if batchWrite.Seqnum == 0 { + batchWrite = trackWriteHandle(t, batchWrite) + if batchWrite.Seqnum() == 0 { t.Fatalf("Write(): Seqnum = 0") } @@ -877,9 +965,12 @@ func TestDbBatchWriteAndConsumption(t *testing.T) { t.Fatalf("WriteBatch.PutWithOptions(): %v", err) } - if _, err := handle.db.WriteWithOptions(secondBatch, slatedb.WriteOptions{AwaitDurable: true}); err != nil { + secondBatchWrite, err := handle.db.WriteWithOptions(secondBatch, slatedb.WriteOptions{AwaitDurable: true, Seqnum: 0}) + if err != nil { t.Fatalf("WriteWithOptions(): %v", err) } + secondBatchWrite = trackWriteHandle(t, secondBatchWrite) + awaitDurable(t, secondBatchWrite) value, err = handle.db.Get([]byte("batch-put-2")) if err != nil { @@ -937,14 +1028,17 @@ func TestDbMerge(t *testing.T) { t.Fatalf("Get(merge) after Merge(): got %v, want %q", value, "base:one") } - if _, err := handle.db.MergeWithOptions( + mergeWrite, err := handle.db.MergeWithOptions( []byte("merge"), []byte(":two"), slatedb.MergeOptions{Ttl: slatedb.TtlDefault{}}, - slatedb.WriteOptions{AwaitDurable: true}, - ); err != nil { + slatedb.WriteOptions{AwaitDurable: true, Seqnum: 0}, + ) + if err != nil { t.Fatalf("MergeWithOptions(): %v", err) } + mergeWrite = trackWriteHandle(t, mergeWrite) + awaitDurable(t, mergeWrite) value, err = handle.db.Get([]byte("merge")) if err != nil { @@ -1024,11 +1118,15 @@ func TestDbTransactions(t *testing.T) { t.Fatalf("db.Get(txn-key) before commit: got %q, want nil", *liveValue) } - commitHandle, err := tx.Commit() + optionalCommitHandle, err := tx.Commit() if err != nil { t.Fatalf("tx.Commit(): %v", err) } - if commitHandle == nil || commitHandle.Seqnum == 0 { + if optionalCommitHandle == nil { + t.Fatal("tx.Commit(): got nil write handle") + } + commitHandle := trackWriteHandle(t, *optionalCommitHandle) + if commitHandle.Seqnum() == 0 { t.Fatalf("tx.Commit(): got %v, want non-nil write handle", commitHandle) } @@ -1124,18 +1222,22 @@ func TestDbInvalidInputsAndErrorMapping(t *testing.T) { t.Fatalf("secondary Put(): %v", err) } - _, err := primary.db.Put([]byte("stale"), []byte("value")) + write, err := primary.db.Put([]byte("stale"), []byte("value")) + if err == nil { + write = trackWriteHandle(t, write) + err = write.AwaitDurable() + } if !errors.Is(err, slatedb.ErrErrorClosed) { - t.Fatalf("primary Put() after fencing: got %v, want closed error", err) + t.Fatalf("primary write durability after fencing: got %v, want closed error", err) } primary.open = false var closedErr *slatedb.ErrorClosed if !errors.As(err, &closedErr) { - t.Fatalf("primary Put() after fencing: expected *ErrorClosed, got %T", err) + t.Fatalf("primary write durability after fencing: expected *ErrorClosed, got %T", err) } if closedErr.Reason != slatedb.CloseReasonFenced { - t.Fatalf("primary Put() after fencing: got close reason %v, want %v", closedErr.Reason, slatedb.CloseReasonFenced) + t.Fatalf("primary write durability after fencing: got close reason %v, want %v", closedErr.Reason, slatedb.CloseReasonFenced) } }) } @@ -1354,6 +1456,7 @@ func TestDbReaderRefreshBehavior(t *testing.T) { t.Helper() if err := builder.WithOptions(slatedb.ReaderOptions{ ManifestPollIntervalMs: 100, + WalPollIntervalMs: 100, CheckpointLifetimeMs: 1000, MaxMemtableBytes: 64 * 1024 * 1024, SkipWalReplay: false, @@ -1386,6 +1489,7 @@ func TestDbReaderWalReplayBehavior(t *testing.T) { t.Helper() if err := builder.WithOptions(slatedb.ReaderOptions{ ManifestPollIntervalMs: 100, + WalPollIntervalMs: 100, CheckpointLifetimeMs: 1000, MaxMemtableBytes: 64 * 1024 * 1024, SkipWalReplay: false, @@ -1432,6 +1536,7 @@ func TestDbReaderWalReplayBehavior(t *testing.T) { t.Helper() if err := builder.WithOptions(slatedb.ReaderOptions{ ManifestPollIntervalMs: 100, + WalPollIntervalMs: 100, CheckpointLifetimeMs: 1000, MaxMemtableBytes: 64 * 1024 * 1024, SkipWalReplay: true, @@ -1464,6 +1569,7 @@ func TestDbReaderWalReplayBehavior(t *testing.T) { t.Helper() if err := builder.WithOptions(slatedb.ReaderOptions{ ManifestPollIntervalMs: 100, + WalPollIntervalMs: 100, CheckpointLifetimeMs: 1000, MaxMemtableBytes: 64 * 1024 * 1024, SkipWalReplay: true, @@ -1778,8 +1884,9 @@ func TestAdminQueries(t *testing.T) { if latestManifest.LastL0Seq < 3 { t.Fatalf("ReadManifest(nil): LastL0Seq = %d, want at least 3", latestManifest.LastL0Seq) } - if latestManifest.WalObjectStoreUri == nil { - t.Fatal("ReadManifest(nil): WalObjectStoreUri = nil, want value for configured WAL store") + // ManifestV2 is now written universally and never persists wal_object_store_uri + if latestManifest.WalObjectStoreUri != nil { + t.Fatalf("ReadManifest(nil): WalObjectStoreUri = %v, want nil (dropped by ManifestV2)", *latestManifest.WalObjectStoreUri) } firstManifest, err := admin.ReadManifest(uint64Ptr(manifests[0].Id)) @@ -2656,207 +2763,166 @@ func TestAdminDeleteMultipleCheckpoints(t *testing.T) { }) } -func TestWalReaderEmptyStore(t *testing.T) { +func flattenWalRows(batches []slatedb.WalRows) []slatedb.RowEntry { + var rows []slatedb.RowEntry + for _, batch := range batches { + rows = append(rows, batch.Rows...) + } + return rows +} + +func TestWalReaderReportsNoNewFilesAfterCursor(t *testing.T) { store := newMemoryStore(t) - reader := openTestWalReader(t, store) + seedWalFiles(t, store) + reader := openTestSlateDbWalReader(t, store) - files, err := reader.List(nil, nil) + cursor, err := reader.LastWalFileId(0) if err != nil { - t.Fatalf("WalReader.List(nil, nil): %v", err) + t.Fatalf("LastWalFileId(0): %v", err) } - for _, file := range files { - defer file.Destroy() + tail, err := reader.LastWalFileId(cursor) + if err != nil { + t.Fatalf("LastWalFileId(cursor): %v", err) } - - if len(files) != 0 { - t.Fatalf("WalReader.List(nil, nil): got %d files, want 0", len(files)) + if tail != cursor { + t.Fatalf("LastWalFileId(cursor): got %d, want %d", tail, cursor) } } -func TestWalReaderListingAndNavigation(t *testing.T) { +func TestWalReaderStreamsNewWalsThroughOneIterator(t *testing.T) { store := newMemoryStore(t) seedWalFiles(t, store) + reader := openTestSlateDbWalReader(t, store) - reader := openTestWalReader(t, store) - - files, err := reader.List(nil, nil) + firstTail, err := reader.LastWalFileId(0) if err != nil { - t.Fatalf("WalReader.List(nil, nil): %v", err) + t.Fatalf("LastWalFileId(0): %v", err) } - - if len(files) < 3 { - t.Fatalf("WalReader.List(nil, nil): got %d files, want at least 3", len(files)) + iter, err := reader.Iterator(1) + if err != nil { + t.Fatalf("SlateDbWalReader.Iterator(1): %v", err) + } + t.Cleanup(iter.Destroy) + firstBatches := readWalBatchesThrough(t, iter, firstTail) + if len(firstBatches) == 0 { + t.Fatal("initial WAL stream returned no batches") } - ids := make([]uint64, len(files)) - for i, file := range files { - ids[i] = file.Id() - if i > 0 && ids[i] <= ids[i-1] { - t.Fatalf("WalReader.List(nil, nil): ids not ascending: %v", ids) + var previous uint64 + foundEmptyFence := false + for i, batch := range firstBatches { + foundEmptyFence = foundEmptyFence || len(batch.Rows) == 0 + if i > 0 && batch.LastConsumedWalFileId <= previous { + t.Fatalf("WAL cursors did not increase: previous=%d current=%d", previous, batch.LastConsumedWalFileId) } + previous = batch.LastConsumedWalFileId } - - startID := ids[1] - endID := ids[2] - bounded, err := reader.List(&startID, &endID) - if err != nil { - t.Fatalf("WalReader.List(start, end): %v", err) + if !foundEmptyFence { + t.Fatal("initial WAL stream did not return the empty fence WAL batch") } - for _, file := range bounded { - defer file.Destroy() + if previous != firstTail { + t.Fatalf("initial WAL stream ended at %d, want %d", previous, firstTail) } - - if len(bounded) != 1 || bounded[0].Id() != ids[1] { - t.Fatalf("WalReader.List(start, end): got ids [%d], want [%d]", len(bounded), ids[1]) + if tail, err := reader.LastWalFileId(firstTail); err != nil || tail != firstTail { + t.Fatalf("LastWalFileId(firstTail): got tail=%d err=%v, want %d", tail, err, firstTail) } - pastHighID := ids[len(ids)-1] + 1000 - empty, err := reader.List(&pastHighID, nil) + appendWalValue(t, store, "next", "3") + secondTail, err := reader.LastWalFileId(firstTail) if err != nil { - t.Fatalf("WalReader.List(pastHigh, nil): %v", err) + t.Fatalf("LastWalFileId(firstTail) after append: %v", err) } - - if len(empty) != 0 { - t.Fatalf("WalReader.List(pastHigh, nil): got %d files, want 0", len(empty)) + if secondTail <= firstTail { + t.Fatalf("second tail did not advance: first=%d second=%d", firstTail, secondTail) } - - first := reader.Get(ids[0]) - defer first.Destroy() - if first.Id() != ids[0] { - t.Fatalf("WalReader.Get(first): got id %d, want %d", first.Id(), ids[0]) + secondBatches := readWalBatchesThrough(t, iter, secondTail) + if len(secondBatches) == 0 { + t.Fatal("continued WAL stream returned no batches") } - if first.NextId() != ids[1] { - t.Fatalf("WalFile.NextId(): got %d, want %d", first.NextId(), ids[1]) + if got := secondBatches[len(secondBatches)-1].LastConsumedWalFileId; got != secondTail { + t.Fatalf("continued WAL stream ended at %d, want %d", got, secondTail) } - - next := first.NextFile() - defer next.Destroy() - if next.Id() != ids[1] { - t.Fatalf("WalFile.NextFile().Id(): got %d, want %d", next.Id(), ids[1]) + secondRows := flattenWalRows(secondBatches) + if len(secondRows) != 1 || string(secondRows[0].Key) != "next" { + t.Fatalf("continued WAL stream returned rows=%v, want one next row", secondRows) + } + if tail, err := reader.LastWalFileId(secondTail); err != nil || tail != secondTail { + t.Fatalf("LastWalFileId(secondTail): got tail=%d err=%v, want %d", tail, err, secondTail) } } -func TestWalReaderMetadataAndRows(t *testing.T) { +func TestWalReaderDecodesValueTombstoneAndMergeRows(t *testing.T) { store := newMemoryStore(t) seedWalFiles(t, store) + reader := openTestSlateDbWalReader(t, store) - reader := openTestWalReader(t, store) - - files, err := reader.List(nil, nil) + tail, err := reader.LastWalFileId(0) if err != nil { - t.Fatalf("WalReader.List(nil, nil): %v", err) - } - for _, file := range files { - defer file.Destroy() - } - - if len(files) < 3 { - t.Fatalf("WalReader.List(nil, nil): got %d files, want at least 3", len(files)) + t.Fatalf("LastWalFileId(0): %v", err) } - - var allRows []slatedb.RowEntry - nonEmptyFiles := 0 - - for i, file := range files { - metadata, err := file.Metadata() - if err != nil { - t.Fatalf("WalFile.Metadata() for file %d: %v", i, err) - } - if metadata.Id != file.Id() { - t.Fatalf("WalFile.Metadata() for file %d: Id = %d, want %d", i, metadata.Id, file.Id()) - } - if metadata.Metadata.Location == "" { - t.Fatalf("WalFile.Metadata() for file %d: Location is empty", i) - } - - iter, err := file.Iterator() - if err != nil { - t.Fatalf("WalFile.Iterator() for file %d: %v", i, err) - } - t.Cleanup(iter.Destroy) - - rows := drainWalIterator(t, iter) - if metadata.Metadata.Size == 0 { - if len(rows) != 0 { - t.Fatalf("zero-byte WAL file %d returned %d rows, want 0", i, len(rows)) - } - continue - } - nonEmptyFiles++ - - for j, row := range rows { - if row.Seq == 0 { - t.Fatalf("row %d in file %d: Seq = 0", j, i) - } - } - allRows = append(allRows, rows...) + iter, err := reader.Iterator(1) + if err != nil { + t.Fatalf("SlateDbWalReader.Iterator(1): %v", err) } - - if nonEmptyFiles == 0 { - t.Fatal("no non-empty WAL files found") + t.Cleanup(iter.Destroy) + batches := readWalBatchesThrough(t, iter, tail) + if got := batches[len(batches)-1].LastConsumedWalFileId; got != tail { + t.Fatalf("WAL stream ended at %d, want %d", got, tail) } - - if len(allRows) != 4 { - t.Fatalf("unexpected total WAL row count: got %d, want 4", len(allRows)) + rows := flattenWalRows(batches) + if len(rows) != 4 { + t.Fatalf("unexpected total WAL row count: got %d, want 4", len(rows)) } - if allRows[0].Kind != slatedb.RowEntryKindValue || string(allRows[0].Key) != "a" { - t.Fatalf("row 0: got kind=%v key=%q, want value/a", allRows[0].Kind, allRows[0].Key) + if rows[0].Kind != slatedb.RowEntryKindValue || string(rows[0].Key) != "a" { + t.Fatalf("row 0: got kind=%v key=%q, want value/a", rows[0].Kind, rows[0].Key) } - if allRows[0].Value == nil || !bytes.Equal(*allRows[0].Value, []byte("1")) { - t.Fatalf("row 0: got value %v, want %q", allRows[0].Value, "1") + if rows[0].Value == nil || !bytes.Equal(*rows[0].Value, []byte("1")) { + t.Fatalf("row 0: got value %v, want %q", rows[0].Value, "1") } - - if allRows[1].Kind != slatedb.RowEntryKindValue || string(allRows[1].Key) != "b" { - t.Fatalf("row 1: got kind=%v key=%q, want value/b", allRows[1].Kind, allRows[1].Key) + if rows[1].Kind != slatedb.RowEntryKindValue || string(rows[1].Key) != "b" { + t.Fatalf("row 1: got kind=%v key=%q, want value/b", rows[1].Kind, rows[1].Key) } - if allRows[1].Value == nil || !bytes.Equal(*allRows[1].Value, []byte("2")) { - t.Fatalf("row 1: got value %v, want %q", allRows[1].Value, "2") + if rows[1].Value == nil || !bytes.Equal(*rows[1].Value, []byte("2")) { + t.Fatalf("row 1: got value %v, want %q", rows[1].Value, "2") } - - if allRows[2].Kind != slatedb.RowEntryKindTombstone || string(allRows[2].Key) != "a" { - t.Fatalf("row 2: got kind=%v key=%q, want tombstone/a", allRows[2].Kind, allRows[2].Key) + if rows[2].Kind != slatedb.RowEntryKindTombstone || string(rows[2].Key) != "a" { + t.Fatalf("row 2: got kind=%v key=%q, want tombstone/a", rows[2].Kind, rows[2].Key) } - if allRows[2].Value != nil { - t.Fatalf("row 2: got value %q, want nil", *allRows[2].Value) + if rows[2].Value != nil { + t.Fatalf("row 2: got value %q, want nil", *rows[2].Value) } - - if allRows[3].Kind != slatedb.RowEntryKindMerge || string(allRows[3].Key) != "m" { - t.Fatalf("row 3: got kind=%v key=%q, want merge/m", allRows[3].Kind, allRows[3].Key) + if rows[3].Kind != slatedb.RowEntryKindMerge || string(rows[3].Key) != "m" { + t.Fatalf("row 3: got kind=%v key=%q, want merge/m", rows[3].Kind, rows[3].Key) } - if allRows[3].Value == nil || !bytes.Equal(*allRows[3].Value, []byte("x")) { - t.Fatalf("row 3: got value %v, want %q", allRows[3].Value, "x") + if rows[3].Value == nil || !bytes.Equal(*rows[3].Value, []byte("x")) { + t.Fatalf("row 3: got value %v, want %q", rows[3].Value, "x") } } -func TestWalReaderMissingFile(t *testing.T) { +func TestWalReaderCanStartAtTheNextWal(t *testing.T) { store := newMemoryStore(t) seedWalFiles(t, store) + reader := openTestSlateDbWalReader(t, store) - reader := openTestWalReader(t, store) - - files, err := reader.List(nil, nil) + tail, err := reader.LastWalFileId(0) if err != nil { - t.Fatalf("WalReader.List(nil, nil): %v", err) - } - for _, file := range files { - defer file.Destroy() + t.Fatalf("LastWalFileId(0): %v", err) } - - if len(files) == 0 { - t.Fatal("WalReader.List(nil, nil): got 0 files, want at least 1") + iter, err := reader.Iterator(tail + 1) + if err != nil { + t.Fatalf("Iterator(next WAL): %v", err) } + t.Cleanup(iter.Destroy) - missingID := files[len(files)-1].Id() + 1000 - missing := reader.Get(missingID) - defer missing.Destroy() - - if missing.Id() != missingID { - t.Fatalf("WalReader.Get(missing): got id %d, want %d", missing.Id(), missingID) + appendWalValue(t, store, "resumed", "4") + newTail, err := reader.LastWalFileId(tail) + if err != nil { + t.Fatalf("LastWalFileId(tail) after append: %v", err) } - - if _, err := missing.Metadata(); err == nil { - t.Fatal("WalFile.Metadata() for missing file: got nil error, want non-nil error") + rows := flattenWalRows(readWalBatchesThrough(t, iter, newTail)) + if len(rows) != 1 || string(rows[0].Key) != "resumed" { + t.Fatalf("resumed WAL stream returned rows=%v, want one resumed row", rows) } } @@ -3072,12 +3138,14 @@ func TestDbTtl(t *testing.T) { key, value := []byte("alpha"), []byte("one") - putOptions := slatedb.PutOptions{Ttl: slatedb.TtlExpireAt{Field0: 1}} - writeOptions := slatedb.WriteOptions{AwaitDurable: true} - _, err := handle.db.PutWithOptions(key, value, putOptions, writeOptions) + putOptions := slatedb.PutOptions{Ttl: slatedb.TtlExpireAtMillis{Field0: 1}} + writeOptions := slatedb.WriteOptions{AwaitDurable: true, Seqnum: 0} + write, err := handle.db.PutWithOptions(key, value, putOptions, writeOptions) if err != nil { t.Fatalf("Put(alpha): %v", err) } + write = trackWriteHandle(t, write) + awaitDurable(t, write) readerHandle := openTestReader(t, store, nil) @@ -3117,26 +3185,29 @@ type batchSeedRow struct { // needs one extra call returning an empty slice to detect exhaustion. var batchSeedRows = []batchSeedRow{ {key: "batch:01", value: "one", ttl: slatedb.TtlNoExpiry{}}, - {key: "batch:02", value: "two", ttl: slatedb.TtlExpireAfterTicks{Field0: batchSeedTtlTicks}}, + {key: "batch:02", value: "two", ttl: slatedb.TtlExpireAfterMillis{Field0: batchSeedTtlMillis}}, {key: "batch:03", value: "three", ttl: slatedb.TtlNoExpiry{}}, - {key: "batch:04", value: "four", ttl: slatedb.TtlExpireAfterTicks{Field0: batchSeedTtlTicks}}, + {key: "batch:04", value: "four", ttl: slatedb.TtlExpireAfterMillis{Field0: batchSeedTtlMillis}}, {key: "batch:05", value: "five", ttl: slatedb.TtlNoExpiry{}}, - {key: "batch:06", value: "six", ttl: slatedb.TtlExpireAfterTicks{Field0: batchSeedTtlTicks}}, + {key: "batch:06", value: "six", ttl: slatedb.TtlExpireAfterMillis{Field0: batchSeedTtlMillis}}, } -// batchSeedTtlTicks is far enough in the future that TTL'd seed rows never +// batchSeedTtlMillis is far enough in the future that TTL'd seed rows never // expire mid-test, while still producing a non-nil ExpireTs. -const batchSeedTtlTicks = 3_600_000 +const batchSeedTtlMillis = 3_600_000 func seedBatchRows(t *testing.T, db *slatedb.Db) { t.Helper() - writeOptions := slatedb.WriteOptions{AwaitDurable: true} + writeOptions := slatedb.WriteOptions{AwaitDurable: true, Seqnum: 0} for _, row := range batchSeedRows { putOptions := slatedb.PutOptions{Ttl: row.ttl} - if _, err := db.PutWithOptions([]byte(row.key), []byte(row.value), putOptions, writeOptions); err != nil { + write, err := db.PutWithOptions([]byte(row.key), []byte(row.value), putOptions, writeOptions) + if err != nil { t.Fatalf("PutWithOptions(%q): %v", row.key, err) } + write = trackWriteHandle(t, write) + awaitDurable(t, write) } } @@ -3387,7 +3458,7 @@ func openBenchDB(b *testing.B) *slatedb.Db { db.Destroy() }) - writeOptions := slatedb.WriteOptions{AwaitDurable: false} + writeOptions := slatedb.WriteOptions{AwaitDurable: false, Seqnum: 0} putOptions := slatedb.PutOptions{Ttl: slatedb.TtlDefault{}} for i := 0; i < benchScanRows; i++ { key := []byte(fmt.Sprintf("bench:%06d", i)) diff --git a/bindings/java/gradle.properties b/bindings/java/gradle.properties index dfb204fc38..baf14210e5 100644 --- a/bindings/java/gradle.properties +++ b/bindings/java/gradle.properties @@ -1 +1 @@ -version=0.15.0-SNAPSHOT +version=0.16.0-SNAPSHOT diff --git a/bindings/java/slatedb-uniffi/src/test/java/io/slatedb/uniffi/SlateDbDbTest.java b/bindings/java/slatedb-uniffi/src/test/java/io/slatedb/uniffi/SlateDbDbTest.java index 8ac8be369b..ae8f3ab5cb 100644 --- a/bindings/java/slatedb-uniffi/src/test/java/io/slatedb/uniffi/SlateDbDbTest.java +++ b/bindings/java/slatedb-uniffi/src/test/java/io/slatedb/uniffi/SlateDbDbTest.java @@ -94,10 +94,11 @@ void dbCrudAndMetadata() throws Exception { ReadOptions readOptions = TestSupport.readOptions(); PutOptions putOptions = new PutOptions(new Ttl.Default()); - WriteOptions writeOptions = new WriteOptions(true); + WriteOptions writeOptions = new WriteOptions(true, 0L); WriteHandle firstWrite = TestSupport.await(db.put(TestSupport.bytes("alpha"), TestSupport.bytes("one"))); assertNotNull(firstWrite); + TestSupport.await(firstWrite.awaitDurable()); assertTrue(firstWrite.seqnum() > 0); assertTrue(firstWrite.createTs() > 0); @@ -126,6 +127,7 @@ void dbCrudAndMetadata() throws Exception { putOptions, writeOptions)); assertNotNull(secondWrite); + TestSupport.await(secondWrite.awaitDurable()); assertTrue(secondWrite.seqnum() > firstWrite.seqnum()); assertTrue(secondWrite.createTs() > 0); @@ -265,7 +267,10 @@ void dbBatchWriteAndConsumption() throws Exception { TestSupport.bytes("value-2"), new PutOptions(new Ttl.Default())); - TestSupport.await(db.writeWithOptions(secondBatch, new WriteOptions(true))); + try (WriteHandle writeHandle = + TestSupport.await(db.writeWithOptions(secondBatch, new WriteOptions(true, 0L)))) { + TestSupport.await(writeHandle.awaitDurable()); + } } assertArrayEquals( @@ -298,12 +303,15 @@ void dbMerge() throws Exception { TestSupport.await(db.merge(TestSupport.bytes("merge"), TestSupport.bytes(":one"))); assertArrayEquals(TestSupport.bytes("base:one"), TestSupport.await(db.get(TestSupport.bytes("merge")))); - TestSupport.await( - db.mergeWithOptions( - TestSupport.bytes("merge"), - TestSupport.bytes(":two"), - new MergeOptions(new Ttl.Default()), - new WriteOptions(true))); + try (WriteHandle writeHandle = + TestSupport.await( + db.mergeWithOptions( + TestSupport.bytes("merge"), + TestSupport.bytes(":two"), + new MergeOptions(new Ttl.Default()), + new WriteOptions(true, 0L)))) { + TestSupport.await(writeHandle.awaitDurable()); + } assertArrayEquals( TestSupport.bytes("base:one:two"), TestSupport.await(db.get(TestSupport.bytes("merge")))); @@ -413,7 +421,15 @@ void dbWriterFencing() throws Exception { Error.Closed error = TestSupport.awaitFailure( Error.Closed.class, - primaryDb.put(TestSupport.bytes("stale"), TestSupport.bytes("value"))); + primaryDb + .put(TestSupport.bytes("stale"), TestSupport.bytes("value")) + .thenCompose( + writeHandle -> + writeHandle + .awaitDurable() + .whenComplete( + (ignored, failure) -> + writeHandle.close()))); primary.markClosed(); assertEquals(CloseReason.FENCED, error.reason()); diff --git a/bindings/java/slatedb-uniffi/src/test/java/io/slatedb/uniffi/SlateDbWalReaderTest.java b/bindings/java/slatedb-uniffi/src/test/java/io/slatedb/uniffi/SlateDbWalReaderTest.java index 979992f0aa..0a65ed75fe 100644 --- a/bindings/java/slatedb-uniffi/src/test/java/io/slatedb/uniffi/SlateDbWalReaderTest.java +++ b/bindings/java/slatedb-uniffi/src/test/java/io/slatedb/uniffi/SlateDbWalReaderTest.java @@ -1,7 +1,7 @@ package io.slatedb.uniffi; import static org.junit.jupiter.api.Assertions.assertEquals; -import static org.junit.jupiter.api.Assertions.assertNotNull; +import static org.junit.jupiter.api.Assertions.assertFalse; import static org.junit.jupiter.api.Assertions.assertTrue; import java.util.ArrayList; @@ -9,137 +9,101 @@ import org.junit.jupiter.api.Test; class SlateDbWalReaderTest { + private static List rowsOf(List batches) { + List rows = new ArrayList<>(); + for (WalRows batch : batches) { + rows.addAll(batch.rows()); + } + return rows; + } + @Test - void walReaderEmptyStore() throws Exception { - try (ObjectStore store = TestSupport.newMemoryStore(); - WalReader reader = TestSupport.openWalReader(store)) { - List files = TestSupport.await(reader.list(null, null)); - try { - assertEquals(0, files.size()); - } finally { - TestSupport.closeAll(files); + void walReaderReportsNoNewFilesAfterCursor() throws Exception { + try (ObjectStore store = TestSupport.newMemoryStore()) { + TestSupport.seedWalFiles(store); + try (SlateDbWalReader reader = TestSupport.openSlateDbWalReader(store)) { + long cursor = TestSupport.await(reader.lastWalFileId(0L)); + assertEquals(cursor, TestSupport.await(reader.lastWalFileId(cursor))); } } } @Test - void walReaderListingAndNavigation() throws Exception { + void walReaderStreamsNewWalsThroughOneIterator() throws Exception { try (ObjectStore store = TestSupport.newMemoryStore()) { TestSupport.seedWalFiles(store); - - try (WalReader reader = TestSupport.openWalReader(store)) { - List files = TestSupport.await(reader.list(null, null)); - try { - assertTrue(files.size() >= 3); - - List ids = new ArrayList<>(); - for (int i = 0; i < files.size(); i++) { - long id = files.get(i).id(); - ids.add(id); - if (i > 0) { - assertTrue(id > ids.get(i - 1)); - } - } - - List bounded = TestSupport.await(reader.list(ids.get(1), ids.get(2))); - try { - assertEquals(1, bounded.size()); - assertEquals(ids.get(1), bounded.get(0).id()); - } finally { - TestSupport.closeAll(bounded); - } - - List empty = TestSupport.await(reader.list(ids.get(ids.size() - 1) + 1000L, null)); - try { - assertEquals(0, empty.size()); - } finally { - TestSupport.closeAll(empty); - } - - try (WalFile first = reader.get(ids.get(0))) { - assertEquals(ids.get(0), first.id()); - assertEquals(ids.get(1), first.nextId()); - - try (WalFile next = first.nextFile()) { - assertEquals(ids.get(1), next.id()); - } + try (SlateDbWalReader reader = TestSupport.openSlateDbWalReader(store)) { + long firstTail = TestSupport.await(reader.lastWalFileId(0L)); + try (SlateDbWalIterator iterator = TestSupport.await(reader.iterator(1L))) { + List firstBatches = + TestSupport.readWalBatchesThrough(iterator, firstTail); + assertFalse(firstBatches.isEmpty()); + assertTrue(firstBatches.stream().anyMatch(batch -> batch.rows().isEmpty())); + long previous = 0L; + for (WalRows batch : firstBatches) { + assertTrue(batch.lastConsumedWalFileId() > previous); + previous = batch.lastConsumedWalFileId(); } - } finally { - TestSupport.closeAll(files); + assertEquals(firstTail, previous); + assertEquals(firstTail, TestSupport.await(reader.lastWalFileId(firstTail))); + + TestSupport.appendWalValue(store, "next", "3"); + long secondTail = TestSupport.await(reader.lastWalFileId(firstTail)); + assertTrue(secondTail > firstTail); + List secondBatches = + TestSupport.readWalBatchesThrough(iterator, secondTail); + assertFalse(secondBatches.isEmpty()); + assertEquals( + secondTail, + secondBatches.get(secondBatches.size() - 1).lastConsumedWalFileId()); + List secondRows = rowsOf(secondBatches); + assertEquals(1, secondRows.size()); + TestSupport.assertWalRow(secondRows.get(0), RowEntryKind.VALUE, "next", "3"); + assertEquals(secondTail, TestSupport.await(reader.lastWalFileId(secondTail))); } } } } @Test - void walReaderMetadataAndRows() throws Exception { + void walReaderDecodesValueTombstoneAndMergeRows() throws Exception { try (ObjectStore store = TestSupport.newMemoryStore()) { TestSupport.seedWalFiles(store); - - try (WalReader reader = TestSupport.openWalReader(store)) { - List files = TestSupport.await(reader.list(null, null)); - try { - assertTrue(files.size() >= 3); - - List allRows = new ArrayList<>(); - int nonEmptyFiles = 0; - for (WalFile file : files) { - IdentifiedObjectMetadata metadata = TestSupport.await(file.metadata()); - assertNotNull(metadata); - assertEquals(file.id(), metadata.id()); - assertFalse(metadata.metadata().location().isEmpty()); - - try (WalFileIterator iterator = TestSupport.await(file.iterator())) { - List rows = TestSupport.drainWalIterator(iterator); - if (metadata.metadata().size() == 0) { - assertTrue(rows.isEmpty()); - continue; - } - - nonEmptyFiles++; - for (RowEntry row : rows) { - assertTrue(row.seq() > 0); - } - allRows.addAll(rows); - } - } - - assertTrue(nonEmptyFiles > 0); - assertEquals(4, allRows.size()); - TestSupport.assertWalRow(allRows.get(0), RowEntryKind.VALUE, "a", "1"); - TestSupport.assertWalRow(allRows.get(1), RowEntryKind.VALUE, "b", "2"); - TestSupport.assertWalRow(allRows.get(2), RowEntryKind.TOMBSTONE, "a", null); - TestSupport.assertWalRow(allRows.get(3), RowEntryKind.MERGE, "m", "x"); - } finally { - TestSupport.closeAll(files); + try (SlateDbWalReader reader = TestSupport.openSlateDbWalReader(store)) { + long tail = TestSupport.await(reader.lastWalFileId(0L)); + List batches; + try (SlateDbWalIterator iterator = TestSupport.await(reader.iterator(1L))) { + batches = TestSupport.readWalBatchesThrough(iterator, tail); + assertEquals(tail, batches.get(batches.size() - 1).lastConsumedWalFileId()); } + + List rows = rowsOf(batches); + assertEquals(4, rows.size()); + assertTrue(rows.stream().allMatch(row -> row.seq() > 0L)); + TestSupport.assertWalRow(rows.get(0), RowEntryKind.VALUE, "a", "1"); + TestSupport.assertWalRow(rows.get(1), RowEntryKind.VALUE, "b", "2"); + TestSupport.assertWalRow(rows.get(2), RowEntryKind.TOMBSTONE, "a", null); + TestSupport.assertWalRow(rows.get(3), RowEntryKind.MERGE, "m", "x"); } } } @Test - void walReaderMissingFile() throws Exception { + void walReaderCanStartAtTheNextWal() throws Exception { try (ObjectStore store = TestSupport.newMemoryStore()) { TestSupport.seedWalFiles(store); - - try (WalReader reader = TestSupport.openWalReader(store)) { - List files = TestSupport.await(reader.list(null, null)); - try { - assertFalse(files.isEmpty()); - long missingId = files.get(files.size() - 1).id() + 1000L; - - try (WalFile missing = reader.get(missingId)) { - assertEquals(missingId, missing.id()); - TestSupport.awaitFailure(Error.class, missing.metadata()); - } - } finally { - TestSupport.closeAll(files); + try (SlateDbWalReader reader = TestSupport.openSlateDbWalReader(store)) { + long tail = TestSupport.await(reader.lastWalFileId(0L)); + try (SlateDbWalIterator iterator = + TestSupport.await(reader.iterator(Math.addExact(tail, 1L)))) { + TestSupport.appendWalValue(store, "resumed", "4"); + long newTail = TestSupport.await(reader.lastWalFileId(tail)); + List rows = rowsOf( + TestSupport.readWalBatchesThrough(iterator, newTail)); + assertEquals(1, rows.size()); + TestSupport.assertWalRow(rows.get(0), RowEntryKind.VALUE, "resumed", "4"); } } } } - - private static void assertFalse(boolean condition) { - org.junit.jupiter.api.Assertions.assertFalse(condition); - } } diff --git a/bindings/java/slatedb-uniffi/src/test/java/io/slatedb/uniffi/TestSupport.java b/bindings/java/slatedb-uniffi/src/test/java/io/slatedb/uniffi/TestSupport.java index 9ffe5af99f..48049306df 100644 --- a/bindings/java/slatedb-uniffi/src/test/java/io/slatedb/uniffi/TestSupport.java +++ b/bindings/java/slatedb-uniffi/src/test/java/io/slatedb/uniffi/TestSupport.java @@ -182,8 +182,8 @@ static ManagedReader openReader(String path, ObjectStore store, DbReaderBuilderC } } - static WalReader openWalReader(ObjectStore store) throws Exception { - return new WalReader(TEST_DB_PATH, store); + static SlateDbWalReader openSlateDbWalReader(ObjectStore store) throws Exception { + return new SlateDbWalReader(TEST_DB_PATH, store); } static void seedWalFiles(ObjectStore store) throws Exception { @@ -201,6 +201,14 @@ static void seedWalFiles(ObjectStore store) throws Exception { } } + static void appendWalValue(ObjectStore store, String key, String value) throws Exception { + try (ManagedDb handle = openDb(store, builder -> builder.withMergeOperator(new ConcatMergeOperator()))) { + Db db = handle.db(); + await(db.put(bytes(key), bytes(value))); + await(db.flushWithOptions(new FlushOptions(FlushType.WAL))); + } + } + static T await(CompletableFuture future) throws Exception { return future.get(TIMEOUT_SECONDS, TimeUnit.SECONDS); } @@ -273,15 +281,18 @@ static List drainIterator(DbIterator iterator) throws Exception { } } - static List drainWalIterator(WalFileIterator iterator) throws Exception { - List rows = new ArrayList<>(); - while (true) { - RowEntry row = await(iterator.next()); - if (row == null) { - return rows; + static List readWalBatchesThrough( + SlateDbWalIterator iterator, long endWalFileId) throws Exception { + List batches = new ArrayList<>(); + while (batches.isEmpty() + || batches.get(batches.size() - 1).lastConsumedWalFileId() < endWalFileId) { + WalRows batch = await(iterator.next()); + if (batch == null) { + throw new AssertionError("live WAL iterator ended unexpectedly"); } - rows.add(row); + batches.add(batch); } + return batches; } static void closeAll(Iterable closeables) throws Exception { @@ -335,7 +346,7 @@ static String uniquePath(String prefix) { } static ReadOptions readOptions() { - return new ReadOptions(DurabilityLevel.MEMORY, false, true, null); + return new ReadOptions(DurabilityLevel.MEMORY, false, true, null, null); } static ScanOptions scanOptions(long readAheadBytes, boolean cacheBlocks, long maxFetchTasks) { @@ -346,11 +357,12 @@ static ScanOptions scanOptions(long readAheadBytes, boolean cacheBlocks, long ma cacheBlocks, maxFetchTasks, null, + null, null); } static ReaderOptions readerOptions(boolean skipWalReplay) { - return new ReaderOptions(100L, 1000L, 64L * 1024 * 1024, skipWalReplay, null); + return new ReaderOptions(100L, 100L, 1000L, 64L * 1024 * 1024, skipWalReplay, null); } private static Throwable unwrap(Throwable thrown) { diff --git a/bindings/node/package.json b/bindings/node/package.json index 91bc2b891c..a35401c454 100644 --- a/bindings/node/package.json +++ b/bindings/node/package.json @@ -1,6 +1,6 @@ { "name": "@slatedb/uniffi", - "version": "0.15.0", + "version": "0.16.0", "description": "Node.js bindings for SlateDB generated from UniFFI and packaged with native libraries.", "license": "Apache-2.0", "type": "module", diff --git a/bindings/node/tests/admin.test.mjs b/bindings/node/tests/admin.test.mjs index 08bf06a27e..8702ac05c7 100644 --- a/bindings/node/tests/admin.test.mjs +++ b/bindings/node/tests/admin.test.mjs @@ -72,26 +72,26 @@ test("admin manifest read list and state view", async (t) => { const admin = openAdmin(store, { path, cleanup }); const db = await openDb(store, { path, cleanup }); - const firstWrite = await db.put_with_options( + const firstWrite = cleanup.track(await db.put_with_options( bytes("alpha"), bytes("one"), putOptions(), writeOptions(false), - ); + )); await db.flush_with_options({ flush_type: FlushType.MemTable }); - const secondWrite = await db.put_with_options( + const secondWrite = cleanup.track(await db.put_with_options( bytes("beta"), bytes("two"), putOptions(), writeOptions(false), - ); + )); await db.flush_with_options({ flush_type: FlushType.MemTable }); const latest = await admin.read_manifest(undefined); assert.notEqual(latest, undefined); assert.ok(BigInt(latest.id) >= 3n); - assert.ok(BigInt(latest.last_l0_seq) >= BigInt(secondWrite.seqnum)); + assert.ok(BigInt(latest.last_l0_seq) >= BigInt(secondWrite.seqnum())); const first = await admin.read_manifest(1n); assert.notEqual(first, undefined); @@ -113,7 +113,7 @@ test("admin manifest read list and state view", async (t) => { const stateView = await admin.read_compactor_state_view(); assert.equal(BigInt(stateView.manifest.id), BigInt(latest.id)); - assert.ok(BigInt(firstWrite.seqnum) > 0n); + assert.ok(BigInt(firstWrite.seqnum()) > 0n); }); test("admin compaction queries handle empty store and invalid ids", async (t) => { @@ -234,27 +234,27 @@ test("admin sequence lookups use persisted tracker", async (t) => { const admin = openAdmin(store, { path, cleanup }); const db = await openDb(store, { path, cleanup }); - const firstWrite = await db.put_with_options( + const firstWrite = cleanup.track(await db.put_with_options( bytes("k1"), bytes("v1"), putOptions(), writeOptions(false), - ); + )); await db.put_with_options( bytes("k2"), bytes("v2"), putOptions(), writeOptions(false), ); - const thirdWrite = await db.put_with_options( + const thirdWrite = cleanup.track(await db.put_with_options( bytes("k3"), bytes("v3"), putOptions(), writeOptions(false), - ); + )); await db.flush_with_options({ flush_type: FlushType.MemTable }); - const firstTimestamp = await admin.get_timestamp_for_sequence(firstWrite.seqnum, true); + const firstTimestamp = await admin.get_timestamp_for_sequence(firstWrite.seqnum(), true); assert.notEqual(firstTimestamp, undefined); const afterLastTimestamp = await admin.get_timestamp_for_sequence(MAX_U64, true); @@ -267,7 +267,7 @@ test("admin sequence lookups use persisted tracker", async (t) => { const seqAfterLast = await admin.get_sequence_for_timestamp(tomorrow, false); assert.notEqual(seqAfterLast, undefined); assert.ok(BigInt(seqAfterLast) > 0n); - assert.ok(BigInt(thirdWrite.seqnum) > BigInt(firstWrite.seqnum)); + assert.ok(BigInt(thirdWrite.seqnum()) > BigInt(firstWrite.seqnum())); const invalidTimestamp = await expectInvalid( () => admin.get_sequence_for_timestamp(MAX_I64, false), diff --git a/bindings/node/tests/db.test.mjs b/bindings/node/tests/db.test.mjs index d7c6b45a78..6e3d30b166 100644 --- a/bindings/node/tests/db.test.mjs +++ b/bindings/node/tests/db.test.mjs @@ -102,14 +102,15 @@ test("db crud and metadata", async (t) => { const store = cleanup.track(newMemoryStore()); const db = await openDb(store, { cleanup }); - const firstWrite = await db.put_with_options( + const firstWrite = cleanup.track(await db.put_with_options( bytes("alpha"), bytes("one"), putOptions(), writeOptions(false), - ); - assert.ok(firstWrite.seqnum > 0); - assert.ok(firstWrite.create_ts > 0); + )); + await firstWrite.await_durable(); + assert.ok(firstWrite.seqnum() > 0); + assert.ok(firstWrite.create_ts() > 0); assert.deepEqual(await db.get(bytes("alpha")), bytes("one")); assert.deepEqual( @@ -121,21 +122,21 @@ test("db crud and metadata", async (t) => { assert.notEqual(metadata, undefined); assert.deepEqual(metadata.key, bytes("alpha")); assert.deepEqual(metadata.value, bytes("one")); - assert.deepEqual(metadata.seq, firstWrite.seqnum); - assert.deepEqual(metadata.create_ts, firstWrite.create_ts); + assert.deepEqual(metadata.seq, firstWrite.seqnum()); + assert.deepEqual(metadata.create_ts, firstWrite.create_ts()); const metadataWithOptions = await db.get_key_value_with_options(bytes("alpha"), readOptions()); assert.notEqual(metadataWithOptions, undefined); assert.deepEqual(metadataWithOptions.value, bytes("one")); - const secondWrite = await db.put_with_options( + const secondWrite = cleanup.track(await db.put_with_options( bytes("beta"), bytes("two"), putOptions(), writeOptions(false), - ); - assert.ok(secondWrite.seqnum > firstWrite.seqnum); - assert.ok(secondWrite.create_ts > 0); + )); + assert.ok(secondWrite.seqnum() > firstWrite.seqnum()); + assert.ok(secondWrite.create_ts() > 0); assert.deepEqual(await db.get(bytes("beta")), bytes("two")); await db.put_with_options( @@ -147,12 +148,12 @@ test("db crud and metadata", async (t) => { assert.deepEqual(await db.get(bytes("empty")), bytes("")); assert.equal(await db.get(bytes("missing")), undefined); - const deleteAlpha = await db.delete_with_options(bytes("alpha"), writeOptions(false)); - assert.ok(deleteAlpha.seqnum > secondWrite.seqnum); + const deleteAlpha = cleanup.track(await db.delete_with_options(bytes("alpha"), writeOptions(false))); + assert.ok(deleteAlpha.seqnum() > secondWrite.seqnum()); assert.equal(await db.get(bytes("alpha")), undefined); - const deleteBeta = await db.delete_with_options(bytes("beta"), writeOptions(false)); - assert.ok(deleteBeta.seqnum > secondWrite.seqnum); + const deleteBeta = cleanup.track(await db.delete_with_options(bytes("beta"), writeOptions(false))); + assert.ok(deleteBeta.seqnum() > secondWrite.seqnum()); assert.equal(await db.get(bytes("beta")), undefined); }); @@ -249,8 +250,8 @@ test("db batch write and consumption", async (t) => { batch.put(bytes("batch-put"), bytes("value")); batch.delete(bytes("remove-me")); - const batchWrite = await db.write(batch); - assert.ok(batchWrite.seqnum > 0); + const batchWrite = cleanup.track(await db.write(batch)); + assert.ok(batchWrite.seqnum() > 0); assert.deepEqual(await db.get(bytes("batch-put")), bytes("value")); assert.equal(await db.get(bytes("remove-me")), undefined); @@ -261,7 +262,8 @@ test("db batch write and consumption", async (t) => { const secondBatch = cleanup.track(new WriteBatch()); secondBatch.put_with_options(bytes("batch-put-2"), bytes("value-2"), putOptions()); - await db.write_with_options(secondBatch, writeOptions()); + const secondBatchWrite = cleanup.track(await db.write_with_options(secondBatch, writeOptions())); + await secondBatchWrite.await_durable(); assert.deepEqual(await db.get(bytes("batch-put-2")), bytes("value-2")); }); @@ -301,12 +303,13 @@ test("db merge and merge_with_options", async (t) => { await db.merge(bytes("merge"), bytes(":one")); assert.deepEqual(await db.get(bytes("merge")), bytes("base:one")); - await db.merge_with_options( + const mergeWrite = cleanup.track(await db.merge_with_options( bytes("merge"), bytes(":two"), mergeOptions(), writeOptions(), - ); + )); + await mergeWrite.await_durable(); assert.deepEqual(await db.get(bytes("merge")), bytes("base:one:two")); }); @@ -346,9 +349,9 @@ test("db transactions", async (t) => { assert.deepEqual(await transaction.get(bytes("txn-key")), bytes("pending")); assert.equal(await db.get(bytes("txn-key")), undefined); - const commitHandle = await transaction.commit(); + const commitHandle = cleanup.track(await transaction.commit()); assert.notEqual(commitHandle, undefined); - assert.ok(commitHandle.seqnum > 0); + assert.ok(commitHandle.seqnum() > 0); assert.deepEqual(await db.get(bytes("txn-key")), bytes("pending")); const rollbackTx = cleanup.track(await db.begin(IsolationLevel.Snapshot)); @@ -423,7 +426,10 @@ test("db writer fencing reports closed reason", async (t) => { ); const error = await expectClosed( - () => primary.put(bytes("stale"), bytes("value")), + async () => { + const write = cleanup.track(await primary.put(bytes("stale"), bytes("value"))); + await write.await_durable(); + }, { reason: CloseReason.Fenced }, ); assert.match(error.message, /detected newer DB client/); diff --git a/bindings/node/tests/support.mjs b/bindings/node/tests/support.mjs index facc730c25..cfba301ad3 100644 --- a/bindings/node/tests/support.mjs +++ b/bindings/node/tests/support.mjs @@ -10,8 +10,8 @@ import { LogLevel, ObjectStore, RowEntryKind, + SlateDbWalReader, Ttl, - WalReader, } from "../index.js"; export const TEST_DB_PATH = "test-db"; @@ -63,6 +63,7 @@ export function scanOptions(readAheadBytes, cacheBlocks, maxFetchTasks) { export function readerOptions(skipWalReplay) { return { manifest_poll_interval_ms: 100, + wal_poll_interval_ms: 100, checkpoint_lifetime_ms: 1_000, max_memtable_bytes: 64 * 1024 * 1024, skip_wal_replay: skipWalReplay, @@ -72,6 +73,7 @@ export function readerOptions(skipWalReplay) { export function writeOptions(awaitDurable = true) { return { await_durable: awaitDurable, + seqnum: 0, }; } @@ -175,8 +177,8 @@ export async function openReader(store, { path = TEST_DB_PATH, configure, cleanu } } -export function openWalReader(store, { path = TEST_DB_PATH, cleanup } = {}) { - const reader = new WalReader(path, store); +export function openSlateDbWalReader(store, { path = TEST_DB_PATH, cleanup } = {}) { + const reader = new SlateDbWalReader(path, store); return cleanup?.track(reader, { shutdown: false }) ?? reader; } @@ -191,15 +193,14 @@ export async function drainIterator(iterator) { } } -export async function drainWalIterator(iterator) { - const rows = []; - for (;;) { - const row = await iterator.next(); - if (row == null) { - return rows; - } - rows.push(row); +export async function readWalBatchesThrough(iterator, endWalFileId) { + const batches = []; + while (batches.length === 0 || BigInt(batches.at(-1).last_consumed_wal_file_id) < endWalFileId) { + const batch = await iterator.next(); + assert.ok(batch != null, "live WAL iterator ended unexpectedly"); + batches.push(batch); } + return batches; } export function requireRows(rows, wantKeys, wantValues) { @@ -364,6 +365,21 @@ export async function seedWalFiles(store) { } } +export async function appendWalValue(store, key, value) { + const db = await openDb(store, { + configure(builder) { + builder.with_merge_operator(new ConcatMergeOperator()); + }, + }); + + try { + await db.put_with_options(bytes(key), bytes(value), putOptions(), writeOptions()); + await db.flush_with_options({ flush_type: FlushType.Wal }); + } finally { + await shutdownAndDispose(db); + } +} + export function uniquePath(prefix) { return `${prefix}-${randomUUID()}`; } diff --git a/bindings/node/tests/wal-reader.test.mjs b/bindings/node/tests/wal-reader.test.mjs index 716957ddf8..40a51f4f5b 100644 --- a/bindings/node/tests/wal-reader.test.mjs +++ b/bindings/node/tests/wal-reader.test.mjs @@ -1,108 +1,96 @@ import assert from "node:assert/strict"; import test from "node:test"; -import { ErrorData } from "../index.js"; import { RowEntryKind, + appendWalValue, createCleanup, - drainWalIterator, - expectError, newMemoryStore, - openWalReader, + openSlateDbWalReader, + readWalBatchesThrough, requireWalRow, seedWalFiles, } from "./support.mjs"; -test("wal reader empty store listing", async (t) => { +function cursorOf(batch) { + return BigInt(batch.last_consumed_wal_file_id); +} + +function rowsOf(batches) { + return batches.flatMap((batch) => batch.rows); +} + +test("wal reader reports no new files after the cursor", async (t) => { const cleanup = createCleanup(t); const store = cleanup.track(newMemoryStore()); - const reader = openWalReader(store, { cleanup }); + await seedWalFiles(store); + const reader = openSlateDbWalReader(store, { cleanup }); - assert.deepEqual(await reader.list(undefined, undefined), []); + const cursor = BigInt(await reader.last_wal_file_id(0n)); + assert.equal(BigInt(await reader.last_wal_file_id(cursor)), cursor); }); -test("wal reader listing bounds and navigation", async (t) => { +test("wal reader streams new WALs through one iterator", async (t) => { const cleanup = createCleanup(t); const store = cleanup.track(newMemoryStore()); await seedWalFiles(store); - - const reader = openWalReader(store, { cleanup }); - const files = (await reader.list(undefined, undefined)).map((file) => cleanup.track(file)); - assert.ok(files.length >= 3); - - const ids = files.map((file) => BigInt(file.id())); + const reader = openSlateDbWalReader(store, { cleanup }); + + const firstTail = BigInt(await reader.last_wal_file_id(0n)); + const iterator = cleanup.track(await reader.iterator(1n)); + const firstBatches = await readWalBatchesThrough(iterator, firstTail); + assert.ok(firstBatches.length > 0); + assert.ok(firstBatches.some((batch) => batch.rows.length === 0)); + const firstCursors = firstBatches.map(cursorOf); + assert.equal(firstCursors.at(-1), firstTail); + assert.ok(firstCursors.every((cursor, index) => index === 0 || cursor > firstCursors[index - 1])); + assert.equal(BigInt(await reader.last_wal_file_id(firstTail)), firstTail); + + await appendWalValue(store, "next", "3"); + const secondTail = BigInt(await reader.last_wal_file_id(firstTail)); + assert.ok(secondTail > firstTail); + const secondBatches = await readWalBatchesThrough(iterator, secondTail); + assert.ok(secondBatches.length > 0); + assert.equal(cursorOf(secondBatches.at(-1)), secondTail); assert.deepEqual( - ids, - [...ids].sort((left, right) => (left < right ? -1 : left > right ? 1 : 0)), + rowsOf(secondBatches).map((row) => Buffer.from(row.key).toString("utf8")), + ["next"], ); - - const bounded = (await reader.list(ids[1], ids[2])).map((file) => cleanup.track(file)); - assert.deepEqual( - bounded.map((file) => BigInt(file.id())), - [ids[1]], - ); - - assert.deepEqual(await reader.list(ids.at(-1) + 1_000n, undefined), []); - - const first = cleanup.track(reader.get(ids[0])); - assert.equal(BigInt(first.id()), ids[0]); - assert.equal(BigInt(first.next_id()), ids[1]); - - const nextFile = cleanup.track(first.next_file()); - assert.equal(BigInt(nextFile.id()), ids[1]); + assert.equal(BigInt(await reader.last_wal_file_id(secondTail)), secondTail); }); -test("wal reader metadata and row decoding", async (t) => { +test("wal reader decodes value, tombstone, and merge rows", async (t) => { const cleanup = createCleanup(t); const store = cleanup.track(newMemoryStore()); await seedWalFiles(store); - - const reader = openWalReader(store, { cleanup }); - const files = (await reader.list(undefined, undefined)).map((file) => cleanup.track(file)); - assert.ok(files.length >= 3); - - const allRows = []; - let nonEmptyFiles = 0; - for (const walFile of files) { - const metadata = await walFile.metadata(); - assert.equal(BigInt(metadata.id), BigInt(walFile.id())); - assert.notEqual(metadata.metadata.location, ""); - - const iterator = cleanup.track(await walFile.iterator()); - const rows = await drainWalIterator(iterator); - if (BigInt(metadata.metadata.size) === 0n) { - assert.deepEqual(rows, []); - continue; - } - - nonEmptyFiles += 1; - for (const row of rows) { - assert.ok(BigInt(row.seq) > 0n); - } - allRows.push(...rows); - } - - assert.ok(nonEmptyFiles > 0); - assert.equal(allRows.length, 4); - requireWalRow(allRows[0], RowEntryKind.Value, "a", "1"); - requireWalRow(allRows[1], RowEntryKind.Value, "b", "2"); - requireWalRow(allRows[2], RowEntryKind.Tombstone, "a", undefined); - requireWalRow(allRows[3], RowEntryKind.Merge, "m", "x"); + const reader = openSlateDbWalReader(store, { cleanup }); + + const tail = BigInt(await reader.last_wal_file_id(0n)); + const batches = await readWalBatchesThrough(cleanup.track(await reader.iterator(1n)), tail); + assert.equal(cursorOf(batches.at(-1)), tail); + const rows = rowsOf(batches); + + assert.equal(rows.length, 4); + assert.ok(rows.every((row) => BigInt(row.seq) > 0n)); + requireWalRow(rows[0], RowEntryKind.Value, "a", "1"); + requireWalRow(rows[1], RowEntryKind.Value, "b", "2"); + requireWalRow(rows[2], RowEntryKind.Tombstone, "a", undefined); + requireWalRow(rows[3], RowEntryKind.Merge, "m", "x"); }); -test("wal reader missing file metadata failure", async (t) => { +test("wal reader can start at the next WAL", async (t) => { const cleanup = createCleanup(t); const store = cleanup.track(newMemoryStore()); await seedWalFiles(store); + const reader = openSlateDbWalReader(store, { cleanup }); + const tail = BigInt(await reader.last_wal_file_id(0n)); - const reader = openWalReader(store, { cleanup }); - const files = (await reader.list(undefined, undefined)).map((file) => cleanup.track(file)); - assert.ok(files.length > 0); - - const missingId = BigInt(files.at(-1).id()) + 1_000n; - const missing = cleanup.track(reader.get(missingId)); - assert.equal(BigInt(missing.id()), missingId); - - const error = await expectError(() => missing.metadata(), ErrorData); - assert.match(error.message, /not found/); + const iterator = cleanup.track(await reader.iterator(tail + 1n)); + await appendWalValue(store, "resumed", "4"); + const newTail = BigInt(await reader.last_wal_file_id(tail)); + const batches = await readWalBatchesThrough(iterator, newTail); + assert.deepEqual( + rowsOf(batches).map((row) => Buffer.from(row.key).toString("utf8")), + ["resumed"], + ); }); diff --git a/bindings/python/tests/conftest.py b/bindings/python/tests/conftest.py index 9ddd245f34..3e45366503 100644 --- a/bindings/python/tests/conftest.py +++ b/bindings/python/tests/conftest.py @@ -29,8 +29,9 @@ RowEntry, RowEntryKind, ScanOptions, + SlateDbWalIterator, Ttl, - WalFileIterator, + WalRows, WriteOptions, ) @@ -52,6 +53,7 @@ def read_options() -> ReadOptions: durability_filter=DurabilityLevel.MEMORY, dirty=False, cache_blocks=True, + tracing_options=None, ) @@ -64,6 +66,7 @@ def scan_options( read_ahead_bytes=read_ahead_bytes, cache_blocks=cache_blocks, max_fetch_tasks=max_fetch_tasks, + tracing_options=None, ) @@ -77,7 +80,7 @@ def reader_options(skip_wal_replay: bool) -> ReaderOptions: def write_options() -> WriteOptions: - return WriteOptions(await_durable=True) + return WriteOptions() def put_options() -> PutOptions: @@ -129,13 +132,15 @@ async def drain_iterator(iterator: DbIterator) -> list[KeyValue]: rows.append(row) -async def drain_wal_iterator(iterator: WalFileIterator) -> list[RowEntry]: - rows: list[RowEntry] = [] - while True: - row = await iterator.next() - if row is None: - return rows - rows.append(row) +async def read_wal_batches_through( + iterator: SlateDbWalIterator, end_wal_file_id: int +) -> list[WalRows]: + batches: list[WalRows] = [] + while not batches or batches[-1].last_consumed_wal_file_id < end_wal_file_id: + batch = await iterator.next() + assert batch is not None, "live WAL iterator ended unexpectedly" + batches.append(batch) + return batches def require_rows(rows: list[KeyValue], want_keys: list[str], want_values: list[str]) -> None: @@ -239,5 +244,14 @@ async def seed_wal_files(store: ObjectStore) -> None: await db.flush_with_options(FlushOptions(flush_type=FlushType.WAL)) +async def append_wal_value(store: ObjectStore, key: bytes, value: bytes) -> None: + async with open_db( + store, + configure=lambda builder: builder.with_merge_operator(ConcatMergeOperator()), + ) as db: + await db.put(key, value) + await db.flush_with_options(FlushOptions(flush_type=FlushType.WAL)) + + def unique_path(prefix: str) -> str: return f"{prefix}-{uuid.uuid4()}" diff --git a/bindings/python/tests/test_admin.py b/bindings/python/tests/test_admin.py index a2ab51df63..add2db69e0 100644 --- a/bindings/python/tests/test_admin.py +++ b/bindings/python/tests/test_admin.py @@ -73,7 +73,7 @@ async def test_admin_manifest_read_list_and_state_view() -> None: latest = await admin.read_manifest(None) assert latest is not None assert latest.id >= 3 - assert latest.last_l0_seq >= second_write.seqnum + assert latest.last_l0_seq >= second_write.seqnum() first = await admin.read_manifest(1) assert first is not None @@ -95,7 +95,7 @@ async def test_admin_manifest_read_list_and_state_view() -> None: state_view = await admin.read_compactor_state_view() assert state_view.manifest.id == latest.id - assert first_write.seqnum > 0 + assert first_write.seqnum() > 0 @pytest.mark.asyncio @@ -200,7 +200,7 @@ async def test_admin_sequence_lookups_use_persisted_tracker() -> None: third_write = await db.put(b"k3", b"v3") await db.flush_with_options(FlushOptions(flush_type=FlushType.MEM_TABLE)) - first_timestamp = await admin.get_timestamp_for_sequence(first_write.seqnum, True) + first_timestamp = await admin.get_timestamp_for_sequence(first_write.seqnum(), True) assert first_timestamp is not None after_last_timestamp = await admin.get_timestamp_for_sequence(MAX_U64, True) @@ -212,7 +212,7 @@ async def test_admin_sequence_lookups_use_persisted_tracker() -> None: seq_after_last = await admin.get_sequence_for_timestamp(int(time.time()) + 86_400, False) assert seq_after_last is not None assert seq_after_last > 0 - assert third_write.seqnum > first_write.seqnum + assert third_write.seqnum() > first_write.seqnum() with pytest.raises(Error.Invalid) as invalid_timestamp: await admin.get_sequence_for_timestamp(MAX_I64, False) diff --git a/bindings/python/tests/test_db.py b/bindings/python/tests/test_db.py index 8d969e123a..36f545aa91 100644 --- a/bindings/python/tests/test_db.py +++ b/bindings/python/tests/test_db.py @@ -96,8 +96,9 @@ async def test_db_crud_and_metadata() -> None: async with open_db(store) as db: first_write = await db.put(b"alpha", b"one") - assert first_write.seqnum > 0 - assert first_write.create_ts > 0 + await first_write.await_durable() + assert first_write.seqnum() > 0 + assert first_write.create_ts() > 0 assert await db.get(b"alpha") == b"one" assert await db.get_with_options(b"alpha", read_options()) == b"one" @@ -106,8 +107,8 @@ async def test_db_crud_and_metadata() -> None: assert metadata is not None assert metadata.key == b"alpha" assert metadata.value == b"one" - assert metadata.seq == first_write.seqnum - assert metadata.create_ts == first_write.create_ts + assert metadata.seq == first_write.seqnum() + assert metadata.create_ts == first_write.create_ts() metadata_with_options = await db.get_key_value_with_options(b"alpha", read_options()) assert metadata_with_options is not None @@ -119,8 +120,9 @@ async def test_db_crud_and_metadata() -> None: put_options(), write_options(), ) - assert second_write.seqnum > first_write.seqnum - assert second_write.create_ts > 0 + await second_write.await_durable() + assert second_write.seqnum() > first_write.seqnum() + assert second_write.create_ts() > 0 assert await db.get(b"beta") == b"two" await db.put(b"empty", b"") @@ -128,11 +130,12 @@ async def test_db_crud_and_metadata() -> None: assert await db.get(b"missing") is None delete_alpha = await db.delete(b"alpha") - assert delete_alpha.seqnum > second_write.seqnum + assert delete_alpha.seqnum() > second_write.seqnum() assert await db.get(b"alpha") is None delete_beta = await db.delete_with_options(b"beta", write_options()) - assert delete_beta.seqnum > second_write.seqnum + await delete_beta.await_durable() + assert delete_beta.seqnum() > second_write.seqnum() assert await db.get(b"beta") is None @@ -248,7 +251,7 @@ async def test_db_batch_write_and_consumption() -> None: batch.delete(b"remove-me") batch_write = await db.write(batch) - assert batch_write.seqnum > 0 + assert batch_write.seqnum() > 0 assert await db.get(b"batch-put") == b"value" assert await db.get(b"remove-me") is None @@ -258,7 +261,8 @@ async def test_db_batch_write_and_consumption() -> None: second_batch = WriteBatch() second_batch.put_with_options(b"batch-put-2", b"value-2", put_options()) - await db.write_with_options(second_batch, write_options()) + second_batch_write = await db.write_with_options(second_batch, write_options()) + await second_batch_write.await_durable() assert await db.get(b"batch-put-2") == b"value-2" @@ -285,12 +289,13 @@ async def test_db_merge_and_merge_with_options() -> None: await db.merge(b"merge", b":one") assert await db.get(b"merge") == b"base:one" - await db.merge_with_options( + merge_write = await db.merge_with_options( b"merge", b":two", merge_options(), write_options(), ) + await merge_write.await_durable() assert await db.get(b"merge") == b"base:one:two" @@ -322,7 +327,7 @@ async def test_db_transactions() -> None: commit_handle = await tx.commit() assert commit_handle is not None - assert commit_handle.seqnum > 0 + assert commit_handle.seqnum() > 0 assert await db.get(b"txn-key") == b"pending" rollback_tx = await db.begin(IsolationLevel.SNAPSHOT) @@ -385,7 +390,8 @@ async def test_db_writer_fencing_reports_closed_reason() -> None: await secondary.put(b"secondary", b"value") with pytest.raises(Error.Closed) as exc: - await primary.put(b"stale", b"value") + write = await primary.put(b"stale", b"value") + await write.await_durable() assert exc.value.reason == CloseReason.FENCED assert "detected newer DB client" in exc.value.message diff --git a/bindings/python/tests/test_wal_reader.py b/bindings/python/tests/test_wal_reader.py index 381d21d09d..1d8ece1118 100644 --- a/bindings/python/tests/test_wal_reader.py +++ b/bindings/python/tests/test_wal_reader.py @@ -3,91 +3,80 @@ import pytest from conftest import ( TEST_DB_PATH, - drain_wal_iterator, + append_wal_value, new_memory_store, + read_wal_batches_through, require_wal_row, seed_wal_files, ) -from slatedb.uniffi import Error, RowEntryKind, WalReader +from slatedb.uniffi import RowEntryKind, SlateDbWalReader @pytest.mark.asyncio -async def test_wal_reader_empty_store_listing() -> None: - reader = WalReader(TEST_DB_PATH, new_memory_store()) - assert await reader.list(None, None) == [] - - -@pytest.mark.asyncio -async def test_wal_reader_listing_bounds_and_navigation() -> None: +async def test_wal_reader_reports_no_new_files_after_cursor() -> None: store = new_memory_store() await seed_wal_files(store) + reader = SlateDbWalReader(TEST_DB_PATH, store) - reader = WalReader(TEST_DB_PATH, store) - files = await reader.list(None, None) - assert len(files) >= 3 - - ids = [wal_file.id() for wal_file in files] - assert ids == sorted(ids) - - bounded = await reader.list(ids[1], ids[2]) - assert [wal_file.id() for wal_file in bounded] == [ids[1]] - - assert await reader.list(ids[-1] + 1_000, None) == [] - - first = reader.get(ids[0]) - assert first.id() == ids[0] - assert first.next_id() == ids[1] - - next_file = first.next_file() - assert next_file.id() == ids[1] + cursor = await reader.last_wal_file_id(0) + assert await reader.last_wal_file_id(cursor) == cursor @pytest.mark.asyncio -async def test_wal_reader_metadata_and_row_decoding() -> None: +async def test_wal_reader_streams_new_wals_through_one_iterator() -> None: store = new_memory_store() await seed_wal_files(store) + reader = SlateDbWalReader(TEST_DB_PATH, store) + + first_tail = await reader.last_wal_file_id(0) + iterator = await reader.iterator(1) + first_batches = await read_wal_batches_through(iterator, first_tail) + assert first_batches + assert any(not batch.rows for batch in first_batches) + first_cursors = [batch.last_consumed_wal_file_id for batch in first_batches] + assert first_cursors[-1] == first_tail + assert first_cursors == sorted(set(first_cursors)) + assert await reader.last_wal_file_id(first_tail) == first_tail + + await append_wal_value(store, b"next", b"3") + second_tail = await reader.last_wal_file_id(first_tail) + assert second_tail > first_tail + second_batches = await read_wal_batches_through(iterator, second_tail) + assert second_batches + assert second_batches[-1].last_consumed_wal_file_id == second_tail + assert [row.key for batch in second_batches for row in batch.rows] == [b"next"] + assert await reader.last_wal_file_id(second_tail) == second_tail - reader = WalReader(TEST_DB_PATH, store) - files = await reader.list(None, None) - assert len(files) >= 3 - all_rows = [] - non_empty_files = 0 - for wal_file in files: - metadata = await wal_file.metadata() - assert metadata.id == wal_file.id() - assert metadata.metadata.location - - rows = await drain_wal_iterator(await wal_file.iterator()) - if metadata.metadata.size == 0: - assert rows == [] - continue +@pytest.mark.asyncio +async def test_wal_reader_decodes_value_tombstone_and_merge_rows() -> None: + store = new_memory_store() + await seed_wal_files(store) + reader = SlateDbWalReader(TEST_DB_PATH, store) - non_empty_files += 1 - assert all(row.seq > 0 for row in rows) - all_rows.extend(rows) + tail = await reader.last_wal_file_id(0) + batches = await read_wal_batches_through(await reader.iterator(1), tail) + assert batches[-1].last_consumed_wal_file_id == tail + rows = [row for batch in batches for row in batch.rows] - assert non_empty_files > 0 - assert len(all_rows) == 4 - require_wal_row(all_rows[0], RowEntryKind.VALUE, "a", "1") - require_wal_row(all_rows[1], RowEntryKind.VALUE, "b", "2") - require_wal_row(all_rows[2], RowEntryKind.TOMBSTONE, "a", None) - require_wal_row(all_rows[3], RowEntryKind.MERGE, "m", "x") + assert len(rows) == 4 + assert all(row.seq > 0 for row in rows) + require_wal_row(rows[0], RowEntryKind.VALUE, "a", "1") + require_wal_row(rows[1], RowEntryKind.VALUE, "b", "2") + require_wal_row(rows[2], RowEntryKind.TOMBSTONE, "a", None) + require_wal_row(rows[3], RowEntryKind.MERGE, "m", "x") @pytest.mark.asyncio -async def test_wal_reader_missing_file_metadata_failure() -> None: +async def test_wal_reader_can_start_at_the_next_wal() -> None: store = new_memory_store() await seed_wal_files(store) - - reader = WalReader(TEST_DB_PATH, store) - files = await reader.list(None, None) - assert files - - missing = reader.get(files[-1].id() + 1_000) - assert missing.id() == files[-1].id() + 1_000 - - with pytest.raises(Error.Data) as exc: - await missing.metadata() - assert "not found" in exc.value.message + reader = SlateDbWalReader(TEST_DB_PATH, store) + tail = await reader.last_wal_file_id(0) + + iterator = await reader.iterator(tail + 1) + await append_wal_value(store, b"resumed", b"4") + new_tail = await reader.last_wal_file_id(tail) + batches = await read_wal_batches_through(iterator, new_tail) + assert [row.key for batch in batches for row in batch.rows] == [b"resumed"] diff --git a/bindings/uniffi/src/admin.rs b/bindings/uniffi/src/admin.rs index ad5d48b9a6..ba0f795278 100644 --- a/bindings/uniffi/src/admin.rs +++ b/bindings/uniffi/src/admin.rs @@ -170,7 +170,7 @@ impl Admin { } /// Deletes the checkpoint with the specified id. - pub async fn delete_checkpoint(&self, id: String) -> Result<(), crate::Error> { + pub async fn delete_checkpoint(&self, id: String) -> Result<(), Error> { self.inner .delete_checkpoint(try_checkpoint_id_from_str(&id)?) .await diff --git a/bindings/uniffi/src/config.rs b/bindings/uniffi/src/config.rs index fcca3fee00..19922307a0 100644 --- a/bindings/uniffi/src/config.rs +++ b/bindings/uniffi/src/config.rs @@ -102,10 +102,10 @@ pub enum Ttl { Default, /// Store the value without expiration. NoExpiry, - /// Expire the value after the given number of clock ticks. - ExpireAfterTicks(u64), - /// Expire the value at the given absolute timestamp (clock ticks). - ExpireAt(i64), + /// Expire the value after the given number of milliseconds. + ExpireAfterMillis(u64), + /// Expire the value at the given Unix timestamp in milliseconds. + ExpireAtMillis(i64), } impl From for slatedb::config::Ttl { @@ -113,8 +113,22 @@ impl From for slatedb::config::Ttl { match value { Ttl::Default => Self::Default, Ttl::NoExpiry => Self::NoExpiry, - Ttl::ExpireAfterTicks(ttl) => Self::ExpireAfter(ttl), - Ttl::ExpireAt(ts) => Self::ExpireAt(ts), + Ttl::ExpireAfterMillis(ttl_millis) => Self::ExpireAfterMillis(ttl_millis), + Ttl::ExpireAtMillis(timestamp_millis) => Self::ExpireAtMillis(timestamp_millis), + } + } +} + +/// Options for tracing a read operation. +#[derive(Clone, Debug, uniffi::Record)] +pub struct TracingOptions { + pub trace_id: String, +} + +impl From for slatedb::config::TracingOptions { + fn from(value: TracingOptions) -> Self { + Self { + trace_id: value.trace_id, } } } @@ -133,6 +147,9 @@ pub struct ReadOptions { /// built-in filters. #[uniffi(default = None)] pub filter_context: Option, + /// Optional caller-supplied tracing settings. + #[uniffi(default = None)] + pub tracing_options: Option, } impl Default for ReadOptions { @@ -142,6 +159,7 @@ impl Default for ReadOptions { dirty: false, cache_blocks: true, filter_context: None, + tracing_options: None, } } } @@ -153,6 +171,7 @@ impl From for slatedb::config::ReadOptions { dirty: value.dirty, cache_blocks: value.cache_blocks, filter_context: value.filter_context.map(Into::into), + tracing_options: value.tracing_options.map(Into::into), } } } @@ -271,6 +290,9 @@ pub struct ScanOptions { /// built-in filters. Only consulted for prefix scans. #[uniffi(default = None)] pub filter_context: Option, + /// Optional caller-supplied tracing settings. + #[uniffi(default = None)] + pub tracing_options: Option, } impl Default for ScanOptions { @@ -283,6 +305,7 @@ impl Default for ScanOptions { max_fetch_tasks: 1, order: None, filter_context: None, + tracing_options: None, } } } @@ -307,21 +330,27 @@ impl TryFrom for slatedb::config::ScanOptions { })?, order: value.order.unwrap_or_default().into(), filter_context: value.filter_context.map(Into::into), + tracing_options: value.tracing_options.map(Into::into), }) } } -/// Options that control durability behavior for writes and commits. +/// Options that control writes and commits. #[derive(Clone, Debug, uniffi::Record)] pub struct WriteOptions { /// Whether the call waits for the write to become durable before returning. + #[uniffi(default = true)] pub await_durable: bool, + /// Optional caller-supplied sequence number. Zero uses SlateDB's sequence oracle. + #[uniffi(default = 0)] + pub seqnum: u64, } impl Default for WriteOptions { fn default() -> Self { Self { await_durable: true, + seqnum: 0, } } } @@ -330,7 +359,7 @@ impl From for slatedb::config::WriteOptions { fn from(value: WriteOptions) -> Self { slatedb::config::WriteOptions { await_durable: value.await_durable, - ..Default::default() + seqnum: value.seqnum, } } } @@ -380,6 +409,30 @@ impl From for slatedb::config::FlushOptions { } } +/// Options controlling how a database is shut down. +#[derive(Clone, Debug, uniffi::Record)] +pub struct CloseOptions { + /// The final flush to perform before shutdown. When `None`, no final flush is + /// triggered and writes that are not durable may be lost. + pub flush_type: Option, +} + +impl Default for CloseOptions { + fn default() -> Self { + Self { + flush_type: Some(FlushType::MemTable), + } + } +} + +impl From for slatedb::config::CloseOptions { + fn from(value: CloseOptions) -> Self { + slatedb::config::CloseOptions { + flush_type: value.flush_type.map(Into::into), + } + } +} + /// Garbage collector options for one age-thresholded directory. #[derive(Clone, Debug, uniffi::Record)] pub struct GarbageCollectorDirectoryOptions { @@ -533,9 +586,39 @@ impl From for slatedb::config::GarbageCollectorOptions #[cfg(test)] mod tests { - use super::{GarbageCollectorOptions, ReaderOptions}; + use super::{CloseOptions, FlushType, GarbageCollectorOptions, ReaderOptions}; use std::time::Duration; + #[test] + fn close_options_default_flushes_memtable() { + let options: slatedb::config::CloseOptions = CloseOptions::default().into(); + + assert!(matches!( + options.flush_type, + Some(slatedb::config::FlushType::MemTable) + )); + } + + #[test] + fn close_options_can_flush_wal_only() { + let options: slatedb::config::CloseOptions = CloseOptions { + flush_type: Some(FlushType::Wal), + } + .into(); + + assert!(matches!( + options.flush_type, + Some(slatedb::config::FlushType::Wal) + )); + } + + #[test] + fn close_options_can_skip_final_flush() { + let options: slatedb::config::CloseOptions = CloseOptions { flush_type: None }.into(); + + assert!(options.flush_type.is_none()); + } + #[test] fn boundary_files_are_enabled_by_default() { let gc: slatedb::config::GarbageCollectorOptions = diff --git a/bindings/uniffi/src/db.rs b/bindings/uniffi/src/db.rs index c8b3b748f9..5ad2d6a635 100644 --- a/bindings/uniffi/src/db.rs +++ b/bindings/uniffi/src/db.rs @@ -1,15 +1,17 @@ use std::sync::Arc; use crate::config::{ - FlushOptions, IsolationLevel, MergeOptions, PutOptions, ReadOptions, ScanOptions, WriteOptions, + CloseOptions, FlushOptions, IsolationLevel, MergeOptions, PutOptions, ReadOptions, ScanOptions, + WriteOptions, }; use crate::db_snapshot::DbSnapshot; use crate::db_transaction::DbTransaction; use crate::error::Error; use crate::iterator::DbIterator; -use crate::types::{CacheTarget, DbStatus, KeyRange, KeyValue, SsTableId, WriteHandle}; +use crate::types::{CacheTarget, DbStatus, KeyRange, KeyValue, SsTableId}; use crate::validation::{validate_key, validate_key_value}; use crate::write_batch::WriteBatch; +use crate::write_handle::WriteHandle; use slatedb::DbCacheManagerOps; /// A writable SlateDB handle. @@ -42,6 +44,15 @@ impl Db { self.inner.close().await.map_err(Into::into) } + /// Performs the requested final flush and closes the database. + #[uniffi::method(name = "shutdown_with_options")] + pub async fn close_with_options(&self, options: CloseOptions) -> Result<(), Error> { + self.inner + .close_with_options(options.into()) + .await + .map_err(Into::into) + } + /// Reads the current value for `key`. pub async fn get(&self, key: Vec) -> Result>, Error> { validate_key(&key)?; @@ -135,9 +146,11 @@ impl Db { /// /// Keys must be non-empty and at most `u16::MAX` bytes. Values must be at /// most `u32::MAX` bytes. - pub async fn put(&self, key: Vec, value: Vec) -> Result { + pub async fn put(&self, key: Vec, value: Vec) -> Result, Error> { validate_key_value(&key, &value)?; - Ok(self.inner.put(key, value).await?.into()) + Ok(Arc::new(WriteHandle::new( + self.inner.put(key, value).await?, + ))) } /// Inserts or overwrites a value using custom put and write options. @@ -147,21 +160,21 @@ impl Db { value: Vec, put_options: PutOptions, write_options: WriteOptions, - ) -> Result { + ) -> Result, Error> { validate_key_value(&key, &value)?; let put_options = put_options.into(); let write_options = write_options.into(); - Ok(self - .inner - .put_with_options(key, value, &put_options, &write_options) - .await? - .into()) + Ok(Arc::new(WriteHandle::new( + self.inner + .put_with_options(key, value, &put_options, &write_options) + .await?, + ))) } /// Deletes `key` and returns metadata for the write. - pub async fn delete(&self, key: Vec) -> Result { + pub async fn delete(&self, key: Vec) -> Result, Error> { validate_key(&key)?; - Ok(self.inner.delete(key).await?.into()) + Ok(Arc::new(WriteHandle::new(self.inner.delete(key).await?))) } /// Deletes `key` using custom write options. @@ -169,16 +182,20 @@ impl Db { &self, key: Vec, options: WriteOptions, - ) -> Result { + ) -> Result, Error> { validate_key(&key)?; let options = options.into(); - Ok(self.inner.delete_with_options(key, &options).await?.into()) + Ok(Arc::new(WriteHandle::new( + self.inner.delete_with_options(key, &options).await?, + ))) } /// Appends a merge operand for `key` and returns metadata for the write. - pub async fn merge(&self, key: Vec, operand: Vec) -> Result { + pub async fn merge(&self, key: Vec, operand: Vec) -> Result, Error> { validate_key_value(&key, &operand)?; - Ok(self.inner.merge(key, operand).await?.into()) + Ok(Arc::new(WriteHandle::new( + self.inner.merge(key, operand).await?, + ))) } /// Appends a merge operand using custom merge and write options. @@ -188,23 +205,23 @@ impl Db { operand: Vec, merge_options: MergeOptions, write_options: WriteOptions, - ) -> Result { + ) -> Result, Error> { validate_key_value(&key, &operand)?; let merge_options = merge_options.into(); let write_options = write_options.into(); - Ok(self - .inner - .merge_with_options(key, operand, &merge_options, &write_options) - .await? - .into()) + Ok(Arc::new(WriteHandle::new( + self.inner + .merge_with_options(key, operand, &merge_options, &write_options) + .await?, + ))) } /// Applies all operations in `batch` atomically. /// /// The provided batch is consumed and cannot be reused afterwards. - pub async fn write(&self, batch: Arc) -> Result { + pub async fn write(&self, batch: Arc) -> Result, Error> { let batch = batch.take_for_write()?; - Ok(self.inner.write(batch).await?.into()) + Ok(Arc::new(WriteHandle::new(self.inner.write(batch).await?))) } /// Applies all operations in `batch` atomically using custom write options. @@ -214,10 +231,12 @@ impl Db { &self, batch: Arc, options: WriteOptions, - ) -> Result { + ) -> Result, Error> { let batch = batch.take_for_write()?; let options = options.into(); - Ok(self.inner.write_with_options(batch, &options).await?.into()) + Ok(Arc::new(WriteHandle::new( + self.inner.write_with_options(batch, &options).await?, + ))) } /// Flushes the default storage layer. diff --git a/bindings/uniffi/src/db_transaction.rs b/bindings/uniffi/src/db_transaction.rs index 338ca115ba..4dad26aec8 100644 --- a/bindings/uniffi/src/db_transaction.rs +++ b/bindings/uniffi/src/db_transaction.rs @@ -5,8 +5,9 @@ use tokio::sync::Mutex; use crate::config::{MergeOptions, PutOptions, ReadOptions, ScanOptions, WriteOptions}; use crate::error::{Error, SlateDbError}; use crate::iterator::DbIterator; -use crate::types::{KeyRange, KeyValue, WriteHandle}; +use crate::types::{KeyRange, KeyValue}; use crate::validation::{validate_key, validate_key_value}; +use crate::write_handle::WriteHandle; /// Transaction handle returned by [`crate::Db::begin`]. /// @@ -226,12 +227,15 @@ impl DbTransaction { /// Commits the transaction. /// /// Returns `None` when the transaction performed no writes. - pub async fn commit(&self) -> Result, Error> { + pub async fn commit(&self) -> Result>, Error> { let tx = { let mut guard = self.inner.lock().await; guard.take().ok_or(SlateDbError::TransactionCompleted)? }; - Ok(tx.commit().await?.map(WriteHandle::from)) + Ok(tx + .commit() + .await? + .map(|handle| Arc::new(WriteHandle::new(handle)))) } /// Commits the transaction using custom write options. @@ -240,7 +244,7 @@ impl DbTransaction { pub async fn commit_with_options( &self, options: WriteOptions, - ) -> Result, Error> { + ) -> Result>, Error> { let options = options.into(); let tx = { let mut guard = self.inner.lock().await; @@ -249,6 +253,6 @@ impl DbTransaction { Ok(tx .commit_with_options(&options) .await? - .map(WriteHandle::from)) + .map(|handle| Arc::new(WriteHandle::new(handle)))) } } diff --git a/bindings/uniffi/src/error.rs b/bindings/uniffi/src/error.rs index b3ab852216..486aaaba86 100644 --- a/bindings/uniffi/src/error.rs +++ b/bindings/uniffi/src/error.rs @@ -149,3 +149,60 @@ impl From for Error { } } } + +impl From for Error { + fn from(error: slatedb::wal::WalError) -> Self { + let message = error.to_string(); + match error { + slatedb::wal::WalError::Fenced => Error::Closed { + reason: CloseReason::Fenced, + message, + }, + slatedb::wal::WalError::Closed => Error::Closed { + reason: CloseReason::Clean, + message, + }, + slatedb::wal::WalError::Unavailable(_) => Error::Unavailable { message }, + slatedb::wal::WalError::WalTruncated(_) | slatedb::wal::WalError::DataError(_) => { + Error::Data { message } + } + slatedb::wal::WalError::InternalError(_) => Error::Internal { message }, + _ => Error::Internal { message }, + } + } +} + +#[cfg(test)] +mod tests { + use std::sync::Arc; + + use super::*; + + #[test] + fn wal_errors_preserve_binding_categories() { + assert!(matches!( + Error::from(slatedb::wal::WalError::WalTruncated(7)), + Error::Data { .. } + )); + assert!(matches!( + Error::from(slatedb::wal::WalError::Unavailable(Arc::new( + std::io::Error::other("offline") + ))), + Error::Unavailable { .. } + )); + assert!(matches!( + Error::from(slatedb::wal::WalError::Fenced), + Error::Closed { + reason: CloseReason::Fenced, + .. + } + )); + assert!(matches!( + Error::from(slatedb::wal::WalError::Closed), + Error::Closed { + reason: CloseReason::Clean, + .. + } + )); + } +} diff --git a/bindings/uniffi/src/lib.rs b/bindings/uniffi/src/lib.rs index 310e4bc947..c746fa147c 100644 --- a/bindings/uniffi/src/lib.rs +++ b/bindings/uniffi/src/lib.rs @@ -19,14 +19,15 @@ mod types; mod validation; mod wal_reader; mod write_batch; +mod write_handle; pub use admin::Admin; pub use builder::{AdminBuilder, CloneBuilder, DbBuilder, DbReaderBuilder}; pub use config::{ - DurabilityLevel, FlushOptions, FlushType, GarbageCollectorDirectoryOptions, + CloseOptions, DurabilityLevel, FlushOptions, FlushType, GarbageCollectorDirectoryOptions, GarbageCollectorOptions, GarbageCollectorScheduleOptions, IsolationLevel, IterationOrder, MergeOptions, PutOptions, ReadOptions, ReaderMode, ReaderOptions, ScanOptions, SstBlockSize, - Ttl, WriteOptions, + TracingOptions, Ttl, WriteOptions, }; pub use db::Db; pub use db_reader::DbReader; @@ -50,9 +51,10 @@ pub use types::{ CompactorStateView, CompressionCodec, DbStatus, ExternalDb, FilterFormat, IdentifiedObjectMetadata, KeyRange, KeyValue, ObjectMetadata, RowEntry, RowEntryKind, Segment, SegmentPrefix, SortedRun, SourceId, SsTableHandle, SsTableId, SsTableInfo, SsTableView, - SstType, VersionedCompactions, VersionedManifest, WriteHandle, + SstType, VersionedCompactions, VersionedManifest, }; -pub use wal_reader::{WalFile, WalFileIterator, WalReader}; +pub use wal_reader::{SlateDbWalIterator, SlateDbWalReader, SlateDbWalReaderOptions, WalRows}; pub use write_batch::WriteBatch; +pub use write_handle::WriteHandle; uniffi::setup_scaffolding!("slatedb"); diff --git a/bindings/uniffi/src/settings.rs b/bindings/uniffi/src/settings.rs index c4fa3f871e..76e13974eb 100644 --- a/bindings/uniffi/src/settings.rs +++ b/bindings/uniffi/src/settings.rs @@ -94,8 +94,8 @@ impl Settings { /// Examples: /// /// - `set("flush_interval", "\"250ms\"")` - /// - `set("default_ttl", "42")` - /// - `set("default_ttl", "null")` + /// - `set("default_ttl_millis", "42")` + /// - `set("default_ttl_millis", "null")` /// - `set("compactor_options.max_sst_size", "33554432")` /// - `set("object_store_cache_options.root_folder", "\"/tmp/slatedb-cache\"")` pub fn set(&self, key: String, value_json: String) -> Result<(), Error> { @@ -169,6 +169,7 @@ fn apply_dotted_json_path(root: &mut Value, key: &str, value: Value) -> Result<( } #[cfg(test)] +#[allow(clippy::result_large_err)] mod tests { use serde_json::json; use std::sync::Arc; @@ -271,13 +272,13 @@ mod tests { let settings = Arc::new(Settings::new(slatedb::Settings::default())); settings - .set("default_ttl".to_owned(), "100".to_owned()) + .set("default_ttl_millis".to_owned(), "100".to_owned()) .unwrap(); settings - .set("default_ttl".to_owned(), "null".to_owned()) + .set("default_ttl_millis".to_owned(), "null".to_owned()) .unwrap(); - assert_eq!(settings.inner().default_ttl, None); + assert_eq!(settings.inner().default_ttl_millis, None); } #[test] @@ -315,13 +316,13 @@ mod tests { let settings = Arc::new(Settings::new(slatedb::Settings::default())); settings - .set("default_ttl".to_owned(), "42".to_owned()) + .set("default_ttl_millis".to_owned(), "42".to_owned()) .unwrap(); let encoded = settings.to_json_string().unwrap(); let decoded = Settings::from_json_string(encoded).unwrap(); - assert_eq!(decoded.inner().default_ttl, Some(42)); + assert_eq!(decoded.inner().default_ttl_millis, Some(42)); } #[test] @@ -367,7 +368,7 @@ flush_interval = "1s" #[test] fn settings_from_env_with_default_uses_default_snapshot() { figment::Jail::expect_with(|jail| { - jail.set_env("FFI_SETTINGS_DEFAULT_TTL", "42"); + jail.set_env("FFI_SETTINGS_DEFAULT_TTL_MILLIS", "42"); let defaults = Arc::new(Settings::new(slatedb::Settings::default())); defaults @@ -382,7 +383,7 @@ flush_interval = "1s" settings.inner().flush_interval, Some(Duration::from_millis(250)) ); - assert_eq!(settings.inner().default_ttl, Some(42)); + assert_eq!(settings.inner().default_ttl_millis, Some(42)); Ok(()) }); diff --git a/bindings/uniffi/src/types.rs b/bindings/uniffi/src/types.rs index 4a33f42be7..b426454915 100644 --- a/bindings/uniffi/src/types.rs +++ b/bindings/uniffi/src/types.rs @@ -108,24 +108,6 @@ impl KeyRange { } } -/// Metadata returned by a successful write. -#[derive(Clone, Debug, PartialEq, Eq, uniffi::Record)] -pub struct WriteHandle { - /// Sequence number assigned to the write. - pub seqnum: u64, - /// Creation timestamp assigned to the write. - pub create_ts: i64, -} - -impl From for WriteHandle { - fn from(value: slatedb::WriteHandle) -> Self { - Self { - seqnum: value.seqnum(), - create_ts: value.create_ts(), - } - } -} - /// A segment (RFC-0024), identified by the key prefix it owns; the segment /// spans the key interval `[prefix, prefix++)`. #[derive(Clone, Debug, PartialEq, Eq, uniffi::Record)] @@ -832,7 +814,7 @@ impl From<&CoreSortedRun> for SortedRun { fn from(value: &CoreSortedRun) -> Self { Self { id: value.id, - sst_views: value.sst_views.iter().map(SsTableView::from).collect(), + sst_views: value.sst_views().iter().map(SsTableView::from).collect(), estimated_size_bytes: value.estimate_size(), } } diff --git a/bindings/uniffi/src/wal_reader.rs b/bindings/uniffi/src/wal_reader.rs index 4148c3ffc5..c7ad872f2b 100644 --- a/bindings/uniffi/src/wal_reader.rs +++ b/bindings/uniffi/src/wal_reader.rs @@ -1,65 +1,85 @@ -use std::ops::Bound; use std::sync::Arc; +use slatedb::wal::WalReader as _; use tokio::sync::Mutex; use crate::error::Error; use crate::object_store::ObjectStore; -use crate::types::{IdentifiedObjectMetadata, RowEntry}; - -/// Handle for a single WAL file. -#[derive(uniffi::Object)] -pub struct WalFile { - inner: slatedb::WalFile, +use crate::types::RowEntry; + +/// Options controlling how the native SlateDB WAL reader fetches WAL SSTs. +#[derive(Clone, Debug, uniffi::Record)] +pub struct SlateDbWalReaderOptions { + /// Number of WAL SSTs to preload. + #[uniffi(default = 4)] + pub sst_batch_size: u64, + /// Number of concurrent fetch tasks per WAL SST. + #[uniffi(default = 2)] + pub max_fetch_tasks: u64, + /// Number of bytes to read ahead from each WAL SST. + #[uniffi(default = 1048576)] + pub read_ahead_bytes: u64, } -impl WalFile { - fn new(inner: slatedb::WalFile) -> Self { - Self { inner } +impl Default for SlateDbWalReaderOptions { + fn default() -> Self { + Self { + sst_batch_size: 4, + max_fetch_tasks: 2, + read_ahead_bytes: 1024 * 1024, + } } } -#[uniffi::export] -impl WalFile { - /// Returns the WAL file ID. - pub fn id(&self) -> u64 { - self.inner.id - } +impl TryFrom for slatedb::wal::SlateDbWalReaderOptions { + type Error = Error; - /// Returns the WAL ID immediately after this file. - pub fn next_id(&self) -> u64 { - self.inner.next_id() + fn try_from(options: SlateDbWalReaderOptions) -> Result { + Ok(Self { + sst_batch_size: positive_usize(options.sst_batch_size, "sst_batch_size")?, + max_fetch_tasks: positive_usize(options.max_fetch_tasks, "max_fetch_tasks")?, + read_ahead_bytes: positive_usize(options.read_ahead_bytes, "read_ahead_bytes")?, + }) } +} - /// Returns a handle for the next WAL file ID without checking existence. - pub fn next_file(&self) -> Arc { - Arc::new(WalFile::new(self.inner.next_file())) +fn positive_usize(value: u64, field: &'static str) -> Result { + if value == 0 { + return Err(Error::Invalid { + message: format!("{field} must be greater than zero"), + }); } + usize::try_from(value).map_err(|_| Error::Invalid { + message: format!("{field} is too large for this platform"), + }) } -#[uniffi::export(async_runtime = "tokio")] -impl WalFile { - /// Reads object-store metadata for this WAL file. - pub async fn metadata(&self) -> Result { - let metadata = self.inner.metadata().await?; - Ok(metadata.into()) - } +/// Rows from one fully consumed WAL file. +#[derive(Clone, Debug, PartialEq, Eq, uniffi::Record)] +pub struct WalRows { + /// Rows stored in the WAL file. Empty fence WALs produce an empty vector. + pub rows: Vec, + /// Last WAL file ID fully consumed by this batch. + pub last_consumed_wal_file_id: u64, +} - /// Opens an iterator over raw row entries in this WAL file. - pub async fn iterator(&self) -> Result, Error> { - let iter = self.inner.iterator().await?; - Ok(Arc::new(WalFileIterator::new(iter))) +impl From for WalRows { + fn from(rows: slatedb::wal::WalRows) -> Self { + Self { + rows: rows.rows.into_iter().map(Into::into).collect(), + last_consumed_wal_file_id: rows.last_consumed_wal_file_id, + } } } -/// Iterator over raw row entries stored in a WAL file. +/// Live iterator over SlateDB WAL files starting at a required WAL file ID. #[derive(uniffi::Object)] -pub struct WalFileIterator { - inner: Mutex, +pub struct SlateDbWalIterator { + inner: Mutex>, } -impl WalFileIterator { - fn new(inner: slatedb::WalFileIterator) -> Self { +impl SlateDbWalIterator { + fn new(inner: Box) -> Self { Self { inner: Mutex::new(inner), } @@ -67,52 +87,116 @@ impl WalFileIterator { } #[uniffi::export(async_runtime = "tokio")] -impl WalFileIterator { - /// Returns the next raw row entry from the WAL file. - pub async fn next(&self) -> Result, Error> { - let mut guard = self.inner.lock().await; - Ok(guard.next().await?.map(RowEntry::from)) +impl SlateDbWalIterator { + /// Returns rows from the next fully consumed WAL file. When it reaches the + /// current tail, this call waits for the next WAL file rather than ending. + pub async fn next(&self) -> Result, Error> { + let mut iterator = self.inner.lock().await; + Ok(iterator.next().await?.map(Into::into)) } } -/// Reader for WAL files stored under a database path. +/// CDC reader backed by SlateDB's native live WAL reader. #[derive(uniffi::Object)] -pub struct WalReader { - inner: slatedb::WalReader, +pub struct SlateDbWalReader { + inner: slatedb::wal::SlateDbWalReader, +} + +impl SlateDbWalReader { + fn build( + path: String, + object_store: Arc, + wal_object_store: Option>, + options: SlateDbWalReaderOptions, + ) -> Result, Error> { + let options = options.try_into()?; + let path = slatedb::object_store::path::Path::from(path); + let mut builder = slatedb::wal::SlateDbWalReaderBuilder::new() + .with_object_store(Arc::clone(&object_store.inner)) + .with_path(path) + .with_options(options); + if let Some(wal_object_store) = wal_object_store { + builder = builder.with_wal_object_store(Arc::clone(&wal_object_store.inner)); + } + let inner = builder.build()?; + Ok(Arc::new(Self { inner })) + } } #[uniffi::export] -impl WalReader { - /// Creates a WAL reader for `path` in `object_store`. +impl SlateDbWalReader { + /// Opens a reader when the manifest and WAL use the same object store. #[uniffi::constructor] - pub fn new(path: String, object_store: Arc) -> Arc { - Arc::new(Self { - inner: slatedb::WalReader::new(path, object_store.inner.clone()), - }) + pub fn new(path: String, object_store: Arc) -> Result, Error> { + Self::build(path, object_store, None, SlateDbWalReaderOptions::default()) + } + + /// Opens a reader with explicit fetch options. + #[uniffi::constructor] + pub fn with_options( + path: String, + object_store: Arc, + options: SlateDbWalReaderOptions, + ) -> Result, Error> { + Self::build(path, object_store, None, options) } - /// Returns a handle for the WAL file with the given ID. - pub fn get(&self, id: u64) -> Arc { - Arc::new(WalFile::new(self.inner.get(id))) + /// Opens a reader for a database with a dedicated WAL object store. + #[uniffi::constructor] + pub fn with_wal_object_store( + path: String, + object_store: Arc, + wal_object_store: Arc, + ) -> Result, Error> { + Self::build( + path, + object_store, + Some(wal_object_store), + SlateDbWalReaderOptions::default(), + ) + } + + /// Opens a reader for a dedicated WAL object store with explicit options. + #[uniffi::constructor] + pub fn with_wal_object_store_and_options( + path: String, + object_store: Arc, + wal_object_store: Arc, + options: SlateDbWalReaderOptions, + ) -> Result, Error> { + Self::build(path, object_store, Some(wal_object_store), options) } } #[uniffi::export(async_runtime = "tokio")] -impl WalReader { - /// Lists WAL files in ascending ID order. - /// - /// `start_id` is inclusive and `end_id` is exclusive when provided. - pub async fn list( - &self, - start_id: Option, - end_id: Option, - ) -> Result>, Error> { - let start = start_id.map(Bound::Included).unwrap_or(Bound::Unbounded); - let end = end_id.map(Bound::Excluded).unwrap_or(Bound::Unbounded); - let files = self.inner.list((start, end)).await?; - Ok(files - .into_iter() - .map(|file| Arc::new(WalFile::new(file))) - .collect()) +impl SlateDbWalReader { + /// Returns a snapshot of the current WAL tail after replay_after_wal_id, or + /// the supplied ID when no later WAL file exists. + pub async fn last_wal_file_id(&self, replay_after_wal_id: u64) -> Result { + Ok(self.inner.last_wal_file_id(replay_after_wal_id).await?) + } + + /// Opens a live iterator starting at start_wal_file_id. The iterator waits + /// and polls internally when it reaches the current WAL tail. + pub async fn iterator(&self, start_wal_file_id: u64) -> Result, Error> { + let iterator = self.inner.iterator((start_wal_file_id..).into()).await?; + Ok(Arc::new(SlateDbWalIterator::new(iterator))) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn rejects_zero_reader_options() { + let options = SlateDbWalReaderOptions { + sst_batch_size: 0, + ..SlateDbWalReaderOptions::default() + }; + assert!(matches!( + slatedb::wal::SlateDbWalReaderOptions::try_from(options), + Err(Error::Invalid { .. }) + )); } } diff --git a/bindings/uniffi/src/write_handle.rs b/bindings/uniffi/src/write_handle.rs new file mode 100644 index 0000000000..cc8d1cdf6d --- /dev/null +++ b/bindings/uniffi/src/write_handle.rs @@ -0,0 +1,34 @@ +use crate::error::Error; + +/// Handle returned by a successful write. +#[derive(uniffi::Object)] +pub struct WriteHandle { + inner: slatedb::WriteHandle, +} + +impl WriteHandle { + pub(crate) fn new(inner: slatedb::WriteHandle) -> Self { + Self { inner } + } +} + +#[uniffi::export] +impl WriteHandle { + /// Returns the sequence number assigned to the write. + pub fn seqnum(&self) -> u64 { + self.inner.seqnum() + } + + /// Returns the creation timestamp assigned to the write. + pub fn create_ts(&self) -> i64 { + self.inner.create_ts() + } +} + +#[uniffi::export(async_runtime = "tokio")] +impl WriteHandle { + /// Waits until the write has been durably persisted. + pub async fn await_durable(&self) -> Result<(), Error> { + self.inner.await_durable().await.map_err(Into::into) + } +} diff --git a/examples/Cargo.toml b/examples/Cargo.toml index 19dc78ad3d..755dd6f2e4 100644 --- a/examples/Cargo.toml +++ b/examples/Cargo.toml @@ -45,6 +45,12 @@ path = "src/refresh_checkpoint.rs" test = false bench = false +[[bin]] +name = "rescaling" +path = "src/rescaling.rs" +test = false +bench = false + [[bin]] name = "range-scans" path = "src/range_scans.rs" diff --git a/examples/src/change_data_capture.rs b/examples/src/change_data_capture.rs index e24566bb4e..b0fc490c01 100644 --- a/examples/src/change_data_capture.rs +++ b/examples/src/change_data_capture.rs @@ -1,14 +1,9 @@ use slatedb::config::{FlushOptions, FlushType}; -use slatedb::object_store::memory::InMemory; -use slatedb::{Db, RowEntry, ValueDeletable, WalFile, WalReader}; +use slatedb::object_store::{memory::InMemory, path::Path}; +use slatedb::wal::{SlateDbWalReaderBuilder, WalReader as _, WalRows}; +use slatedb::{Db, RowEntry, ValueDeletable}; use std::sync::Arc; -#[derive(Debug, Default)] -struct CdcCursor { - wal_id: u64, - last_seq: u64, -} - #[tokio::main] async fn main() -> anyhow::Result<()> { let object_store = Arc::new(InMemory::new()); @@ -18,41 +13,62 @@ async fn main() -> anyhow::Result<()> { db.put(b"user:1", b"alice").await?; db.put(b"user:2", b"bob").await?; db.delete(b"user:2").await?; - db.flush_with_options(FlushOptions { - flush_type: FlushType::Wal, - }) - .await?; + flush_wal(&db).await?; - let wal_reader = WalReader::new(path, object_store); - let mut cursor = CdcCursor::default(); + let wal_reader = SlateDbWalReaderBuilder::new() + .with_object_store(object_store) + .with_path(Path::from(path)) + .build()?; + let mut cursor = 0_u64; + let start_wal_id = cursor + .checked_add(1) + .ok_or_else(|| anyhow::anyhow!("WAL cursor cannot advance"))?; + let mut iterator = wal_reader.iterator((start_wal_id..).into()).await?; - // Use list() once for discovery (or after a long outage). - for wal_file in wal_reader.list(cursor.wal_id..).await? { - emit_wal_file(&wal_file, &mut cursor).await?; + // Drain the writes that already exist. Empty fence WALs still advance the + // cursor, so stop based on emitted rows only after persisting every batch. + let mut emitted_rows = 0; + while emitted_rows < 3 { + let batch = iterator + .next() + .await? + .ok_or_else(|| anyhow::anyhow!("live WAL iterator ended unexpectedly"))?; + emitted_rows += emit_batch(&batch, &mut cursor); } - // Poll by ID to avoid repeated full prefix listings. - let next_file = wal_reader.get(cursor.wal_id + 1); - emit_wal_file(&next_file, &mut cursor).await?; - println!("Persist cursor periodically: {:?}", cursor); + // Keep the same iterator alive. Its next call observes this later WAL; + // callers do not need to discover a new tail or create another iterator. + db.put(b"user:3", b"carol").await?; + flush_wal(&db).await?; + while emitted_rows < 4 { + let batch = iterator + .next() + .await? + .ok_or_else(|| anyhow::anyhow!("live WAL iterator ended unexpectedly"))?; + emitted_rows += emit_batch(&batch, &mut cursor); + } db.close().await?; Ok(()) } -async fn emit_wal_file(wal_file: &WalFile, cursor: &mut CdcCursor) -> anyhow::Result<()> { - let mut iter = wal_file.iterator().await?; - while let Some(row) = iter.next().await? { - if wal_file.id == cursor.wal_id && row.seq <= cursor.last_seq { - continue; - } +async fn flush_wal(db: &Db) -> Result<(), slatedb::Error> { + db.flush_with_options(FlushOptions { + flush_type: FlushType::Wal, + }) + .await +} - emit_row(wal_file.id, &row); - cursor.wal_id = wal_file.id; - cursor.last_seq = row.seq; +fn emit_batch(batch: &WalRows, cursor: &mut u64) -> usize { + for row in &batch.rows { + emit_row(batch.last_consumed_wal_file_id, row); } - Ok(()) + // Persist only after every row in this WAL file has been emitted. This + // also advances across empty fence WALs. + *cursor = batch.last_consumed_wal_file_id; + println!("persist cursor={cursor}"); + batch.rows.len() } fn emit_row(wal_id: u64, row: &RowEntry) { diff --git a/examples/src/create_snapshot.rs b/examples/src/create_snapshot.rs index 982ea360d8..29c3b7c89d 100644 --- a/examples/src/create_snapshot.rs +++ b/examples/src/create_snapshot.rs @@ -5,7 +5,7 @@ use std::sync::Arc; #[tokio::main] async fn main() -> Result<(), Error> { // Initialize database - let object_store = Arc::new(slatedb::object_store::memory::InMemory::new()); + let object_store = Arc::new(object_store::memory::InMemory::new()); let db = Db::builder("my_db", object_store) .with_settings(Settings::default()) .build() diff --git a/examples/src/rescaling.rs b/examples/src/rescaling.rs new file mode 100644 index 0000000000..663968196d --- /dev/null +++ b/examples/src/rescaling.rs @@ -0,0 +1,356 @@ +//! Rescaling a database by splitting and merging key ranges. +//! +//! Scale-up (split) projects one source into two clones with disjoint key +//! ranges. Scale-down (merge) unions those clones back into one database. +//! Both are O(1) manifest views over shared SSTs — no SST data is copied. +//! +//! Projection and union reject sources that still have data in the WAL, so +//! this example flushes the WAL and memtables into L0, then checkpoints with +//! [`CheckpointScope::Durable`] before cloning. The same APIs work for +//! segmented and non-segmented stores. +//! +//! Union requires its sources to be non-overlapping. For a segmented store +//! that rule applies per segment, so the two shards below merge in one call +//! even though each holds part of both segments. +//! +//! Clone construction uses: +//! [`AdminBuilder::new`] → [`Admin::create_clone_builder_from_source`] → +//! [`CloneBuilder::with_source`] → [`CloneBuilder::build`]. + +use slatedb::admin::{AdminBuilder, CloneSourceSpec}; +use slatedb::bytes::Bytes; +use slatedb::config::{CheckpointOptions, CheckpointScope, FlushOptions, FlushType}; +use slatedb::object_store::memory::InMemory; +use slatedb::{CheckpointCreateResult, Db, Error, PrefixExtractor, PrefixTarget}; +use std::ops::{Bound, RangeBounds}; +use std::sync::Arc; + +type ProjectionRange = (Bound, Bound); + +/// Tenant IDs sort as bytes, so zoos `< "metro"` land on the left shard. +const SPLIT_TENANT: &[u8] = b"metro"; + +fn left_tenants() -> ProjectionRange { + ( + Bound::Unbounded, + Bound::Excluded(Bytes::from_static(SPLIT_TENANT)), + ) +} + +fn right_tenants() -> ProjectionRange { + ( + Bound::Included(Bytes::from_static(SPLIT_TENANT)), + Bound::Unbounded, + ) +} + +/// Two LSM segments — bulky animal records vs a smaller owner index. +/// +/// Keys are kind-first (`data/…`, `idx/…`) so the extractor names the segments +/// `data` and `idx`. Tenants live in the next path component, so each zoo's +/// rows are not one contiguous byte range; tenant splits use +/// [`CloneBuilder::with_segment_projection`]. +struct DataIdxSegmentExtractor; + +impl PrefixExtractor for DataIdxSegmentExtractor { + fn name(&self) -> &str { + "data_idx" + } + + fn prefix_len(&self, target: &PrefixTarget) -> Option { + let key = match target { + PrefixTarget::Point(key) | PrefixTarget::Prefix(key) => key.as_ref(), + }; + if key == b"data" || key.starts_with(b"data/") { + Some(b"data".len()) + } else if key == b"idx" || key.starts_with(b"idx/") { + Some(b"idx".len()) + } else { + None + } + } +} + +fn animal_key(zoo: &[u8], animal_id: &[u8]) -> Vec { + [b"data/", zoo, b"/animal/", animal_id].concat() +} + +fn owner_index_key(zoo: &[u8], owner: &[u8], animal_id: &[u8]) -> Vec { + [b"idx/", zoo, b"/owner/", owner, b"/", animal_id].concat() +} + +fn kind_tenant(kind: &[u8], tenant: &[u8]) -> Bytes { + Bytes::from([kind, b"/", tenant].concat()) +} + +/// Per-segment view of zoos `< metro` (valid inside `[prefix, prefix++)`). +fn left_tenant_in_segment(prefix: &[u8]) -> ProjectionRange { + ( + Bound::Unbounded, + Bound::Excluded(kind_tenant(prefix, SPLIT_TENANT)), + ) +} + +/// Per-segment view of zoos `>= metro`. +fn right_tenant_in_segment(prefix: &[u8]) -> ProjectionRange { + ( + Bound::Included(kind_tenant(prefix, SPLIT_TENANT)), + Bound::Unbounded, + ) +} + +#[tokio::main] +async fn main() -> anyhow::Result<()> { + let object_store = Arc::new(InMemory::new()); + + println!("=== Non-segmented rescaling ==="); + rescale_non_segmented(object_store.clone()).await?; + + println!("\n=== Segmented rescaling (data + idx segments, split by zoo) ==="); + rescale_segmented(object_store).await?; + + Ok(()) +} + +async fn rescale_non_segmented(object_store: Arc) -> anyhow::Result<()> { + let root_path = "/tmp/slatedb_rescaling/plain/root"; + let left_path = "/tmp/slatedb_rescaling/plain/left"; + let right_path = "/tmp/slatedb_rescaling/plain/right"; + let merged_path = "/tmp/slatedb_rescaling/plain/merged"; + + // Tenant-prefixed keys without a segment extractor — still split by zoo. + let db = Db::open(root_path, object_store.clone()).await?; + db.put(b"bronx/lion", b"Leo").await?; + db.put(b"lincoln/otter", b"Ollie").await?; + db.put(b"metro/panda", b"Mei").await?; + db.put(b"oakland/zebra", b"Ziggy").await?; + let checkpoint = checkpoint_for_rescale(&db).await?; + db.close().await?; + + create_clone( + left_path, + vec![CloneSourceSpec::with_checkpoint(root_path, checkpoint.id) + .with_projection_range(left_tenants())], + object_store.clone(), + ) + .await?; + create_clone( + right_path, + vec![CloneSourceSpec::with_checkpoint(root_path, checkpoint.id) + .with_projection_range(right_tenants())], + object_store.clone(), + ) + .await?; + + let left = Db::open(left_path, object_store.clone()).await?; + let right = Db::open(right_path, object_store.clone()).await?; + assert_eq!( + left.get(b"bronx/lion").await?, + Some(b"Leo".as_slice().into()) + ); + assert_eq!(left.get(b"metro/panda").await?, None); + assert_eq!( + right.get(b"metro/panda").await?, + Some(b"Mei".as_slice().into()) + ); + assert_eq!(right.get(b"bronx/lion").await?, None); + println!("split by tenant: left has bronx/lincoln; right has metro/oakland"); + left.close().await?; + right.close().await?; + + create_clone( + merged_path, + vec![ + CloneSourceSpec::new(left_path).with_projection_range(left_tenants()), + CloneSourceSpec::new(right_path).with_projection_range(right_tenants()), + ], + object_store.clone(), + ) + .await?; + + let merged = Db::open(merged_path, object_store).await?; + assert_eq!( + merged.get(b"bronx/lion").await?, + Some(b"Leo".as_slice().into()) + ); + assert_eq!( + merged.get(b"oakland/zebra").await?, + Some(b"Ziggy".as_slice().into()) + ); + println!("merged: all zoos are visible again"); + merged.close().await?; + + Ok(()) +} + +async fn rescale_segmented(object_store: Arc) -> anyhow::Result<()> { + let extractor = Arc::new(DataIdxSegmentExtractor); + let root_path = "/tmp/slatedb_rescaling/segmented/root"; + let left_path = "/tmp/slatedb_rescaling/segmented/left"; + let right_path = "/tmp/slatedb_rescaling/segmented/right"; + let merged_path = "/tmp/slatedb_rescaling/segmented/merged"; + + let db = Db::builder(root_path, object_store.clone()) + .with_segment_extractor(extractor.clone()) + .build() + .await?; + + // bronx + lincoln → left of the split; metro + oakland → right. + put_animal(&db, b"bronx", b"lion-1", b"alice", b"Leo the lion").await?; + put_animal(&db, b"lincoln", b"otter-1", b"bob", b"Ollie the otter").await?; + put_animal(&db, b"metro", b"panda-1", b"carol", b"Mei the panda").await?; + put_animal(&db, b"oakland", b"zebra-1", b"dave", b"Ziggy the zebra").await?; + + let checkpoint = checkpoint_for_rescale(&db).await?; + db.close().await?; + + // Scale up: keep each zoo's data + owner-index together via per-segment + // projection (`data/{zoo}/…` and `idx/{zoo}/…` are not one byte range). + create_clone_with_segment_projection( + left_path, + CloneSourceSpec::with_checkpoint(root_path, checkpoint.id), + object_store.clone(), + left_tenant_in_segment, + ) + .await?; + create_clone_with_segment_projection( + right_path, + CloneSourceSpec::with_checkpoint(root_path, checkpoint.id), + object_store.clone(), + right_tenant_in_segment, + ) + .await?; + + let left = Db::builder(left_path, object_store.clone()) + .with_segment_extractor(extractor.clone()) + .build() + .await?; + let right = Db::builder(right_path, object_store.clone()) + .with_segment_extractor(extractor.clone()) + .build() + .await?; + + assert_eq!( + left.get(animal_key(b"bronx", b"lion-1")).await?, + Some(b"Leo the lion".as_slice().into()) + ); + assert_eq!( + left.get(owner_index_key(b"bronx", b"alice", b"lion-1")) + .await?, + Some(Bytes::new()) + ); + assert_eq!(left.get(animal_key(b"metro", b"panda-1")).await?, None); + assert_eq!( + left.get(owner_index_key(b"metro", b"carol", b"panda-1")) + .await?, + None + ); + assert_eq!( + right.get(animal_key(b"metro", b"panda-1")).await?, + Some(b"Mei the panda".as_slice().into()) + ); + assert_eq!( + right + .get(owner_index_key(b"metro", b"carol", b"panda-1")) + .await?, + Some(Bytes::new()) + ); + assert_eq!(right.get(animal_key(b"bronx", b"lion-1")).await?, None); + println!("split by tenant: left has bronx/lincoln (data+idx); right has metro/oakland"); + left.close().await?; + right.close().await?; + + // Scale down: one union merges both shards. Union requires the sources to + // be non-overlapping per segment, not overall — each shard holds the lower + // or upper zoos of both `data` and `idx`, so no segment is claimed twice. + create_clone( + merged_path, + vec![ + CloneSourceSpec::new(left_path), + CloneSourceSpec::new(right_path), + ], + object_store.clone(), + ) + .await?; + + let merged = Db::builder(merged_path, object_store) + .with_segment_extractor(extractor) + .build() + .await?; + assert_eq!( + merged.get(animal_key(b"bronx", b"lion-1")).await?, + Some(b"Leo the lion".as_slice().into()) + ); + assert_eq!( + merged + .get(owner_index_key(b"oakland", b"dave", b"zebra-1")) + .await?, + Some(Bytes::new()) + ); + println!("merged: all zoo data and owner-index rows are visible again"); + merged.close().await?; + + Ok(()) +} + +async fn put_animal( + db: &Db, + zoo: &[u8], + animal_id: &[u8], + owner: &[u8], + record: &[u8], +) -> Result<(), Error> { + db.put(animal_key(zoo, animal_id), record).await?; + db.put(owner_index_key(zoo, owner, animal_id), b"").await?; + Ok(()) +} + +/// Build a clone from one or more sources. +async fn create_clone( + clone_path: &str, + sources: Vec>, + object_store: Arc, +) -> Result<(), Error> { + let admin = AdminBuilder::new(clone_path, object_store).build(); + let mut sources = sources.into_iter(); + let first = sources + .next() + .expect("rescaling clone requires at least one source"); + let mut builder = admin.create_clone_builder_from_source(first); + for source in sources { + builder = builder.with_source(source); + } + builder.build().await +} + +/// Build a single-source clone with a per-segment projection. +async fn create_clone_with_segment_projection( + clone_path: &str, + source: CloneSourceSpec, + object_store: Arc, + segment_projection: F, +) -> Result<(), Error> +where + F: Fn(&[u8]) -> R + Send + Sync + 'static, + R: RangeBounds, +{ + AdminBuilder::new(clone_path, object_store) + .build() + .create_clone_builder_from_source(source) + .with_segment_projection(segment_projection) + .build() + .await +} + +/// Flush every write into L0, then pin that state. Projection and union reject +/// sources that still have non-empty WAL SSTs. +async fn checkpoint_for_rescale(db: &Db) -> anyhow::Result { + db.flush().await?; + db.flush_with_options(FlushOptions { + flush_type: FlushType::MemTable, + }) + .await?; + Ok(db + .create_checkpoint(CheckpointScope::Durable, &CheckpointOptions::default()) + .await?) +} diff --git a/examples/src/scan_snapshot.rs b/examples/src/scan_snapshot.rs index 9d4441d01e..706702697b 100644 --- a/examples/src/scan_snapshot.rs +++ b/examples/src/scan_snapshot.rs @@ -5,7 +5,7 @@ use std::sync::Arc; #[tokio::main] async fn main() -> Result<(), Error> { // Initialize database - let object_store = Arc::new(slatedb::object_store::memory::InMemory::new()); + let object_store = Arc::new(object_store::memory::InMemory::new()); let db = Db::builder("my_db", object_store) .with_settings(Settings::default()) .build() diff --git a/rfcs/0003-timestamps-and-ttl.md b/rfcs/0003-timestamps-and-ttl.md index f63b9bee48..0262279836 100644 --- a/rfcs/0003-timestamps-and-ttl.md +++ b/rfcs/0003-timestamps-and-ttl.md @@ -93,11 +93,11 @@ tick and sleeping briefly if clock skew is detected. pub struct DbOptions { // ... - /// The default time-to-live (TTL) for insertions (note that re-inserting a key - /// with any value will update the TTL to use the default_ttl) + /// The default time-to-live (TTL), in milliseconds, for insertions (note that + /// re-inserting a key with any value will update the TTL to use default_ttl_millis) /// /// Default: no TTL (insertions will remain until deleted) - default_ttl: Option + default_ttl_millis: Option } ``` @@ -126,7 +126,7 @@ pub enum Ttl { /// No expiration for this entry NoExpiry, /// Expire after the specified duration (in milliseconds) - ExpireAfter(u64), + ExpireAfterMillis(u64), } pub struct PutOptions { @@ -367,4 +367,4 @@ the original `seq0` insert as it logically happened "after" `seq1`. for testing or non-standard time sources. - Updated `DbOptions` to remove the `clock` configuration option - Updated `WriteOptions` to `PutOptions` with a `Ttl` enum that supports `Default`, `NoExpiry`, - and `ExpireAfter(u64)` variants + and `ExpireAfterMillis(u64)` variants diff --git a/rfcs/0004-checkpoints.md b/rfcs/0004-checkpoints.md index 69868e0977..d6fd4eac71 100644 --- a/rfcs/0004-checkpoints.md +++ b/rfcs/0004-checkpoints.md @@ -748,6 +748,8 @@ The union process works as follows: of all its L0 and compacted SSTs and optionally intersected with `visible_range`s for each manifest if they are provided by the user. If any two manifests have intersecting key ranges, the operation fails. If the `visible_ranges` are explicitly provided then validate that they are adjacent. + Segmented manifests apply this check per segment rather than to the manifest as a whole; see + [RFC 0024](./0024-segment-oriented-compaction.md#interaction-with-projection-and-union). 3. Merge the contents of all input manifests: - `external_dbs` entries from all input manifests are merged and deduplicated by `(path, source_checkpoint_id)`. `external_dbs` with the same `(path, source_checkpoint_id)` originated from the diff --git a/rfcs/0006-merge-operator.md b/rfcs/0006-merge-operator.md index 1bf4b1f323..d3f873dffc 100644 --- a/rfcs/0006-merge-operator.md +++ b/rfcs/0006-merge-operator.md @@ -220,12 +220,12 @@ impl DbOptions { ### Extending TTL support -The last public API change is extending the `Ttl` enum to support a new `ExpireAt(ts)` variant. This allows users to set a specific expiration time for a key, which overrides the default TTL behavior. +The last public API change is extending the `Ttl` enum to support a new `ExpireAtMillis(timestamp_millis)` variant. This allows users to set a specific expiration time for a key, expressed as milliseconds since the Unix epoch, which overrides the default TTL behavior. ```rust pub enum Ttl { ... - ExpireAt(i64), + ExpireAtMillis(i64), } ``` @@ -425,8 +425,8 @@ SlateDB supports two TTL approaches: 1. **Operation-Level TTL** - Each operation (put/merge) has its own independent TTL, specified via: - - `Ttl::ExpireAfter(duration)`: Expires after specified duration (internally this is implemented as `ExpireAt(Instant::now() + duration)`) - - `Ttl::ExpireAt(timestamp)`: Expires at specified timestamp + - `Ttl::ExpireAfterMillis(duration_millis)`: Expires after the specified duration in milliseconds (internally this is implemented as `create_ts + duration_millis`) + - `Ttl::ExpireAtMillis(timestamp_millis)`: Expires at the specified Unix timestamp in milliseconds - Enables per-element expiration in collections 2. **TTL Renewal (NOT SUPPORTED NATIVELY)** @@ -439,7 +439,7 @@ When merging values with different TTLs, the merge operation only combines value Unlike regular values which become tombstones upon expiration, expired merge entries are simply removed, enabling per-element expiration in collections. -Users can implement custom TTL patterns by consistently using either `ExpireAt` or `ExpireAfter` across operations. +Users can implement custom TTL patterns by consistently using either `ExpireAtMillis` or `ExpireAfterMillis` across operations. ### Ordering Guarantees @@ -498,4 +498,4 @@ One possible optimization is to introduce a new read option for persisting the r It might be useful to have a read option that allows for early termination of merge operations. This could be beneficial for buffering use cases where the user wants to limit the number of operands that are merged together (effectively allowing partial iterations). -RocksDB provides an alternative approach by exposing `GetMergeOperands` which allows listing the unmerged operands directly. \ No newline at end of file +RocksDB provides an alternative approach by exposing `GetMergeOperands` which allows listing the unmerged operands directly. diff --git a/rfcs/0020-range-metadata.md b/rfcs/0020-range-metadata.md index 4563f423a0..6b1d266047 100644 --- a/rfcs/0020-range-metadata.md +++ b/rfcs/0020-range-metadata.md @@ -153,11 +153,16 @@ impl SstFile { /// SSTs that were written before the stats block was added. pub async fn stats(&self) -> Result, crate::Error>; - /// Returns `(block_offset, first_key)` pairs from the SST index block. - pub async fn index(&self) -> Result, crate::Error>; + /// Returns a zero-copy view of the SST index block. + pub async fn index(&self) -> Result; } ``` +`SstIndex` retains the cached index data and provides indexed access, +iteration, and binary-search partition points over borrowed first keys. This +avoids materializing and copying the full index for consumers that only need +to locate blocks. + ```rust pub struct SstStats { pub num_puts: u64, @@ -181,7 +186,7 @@ The `SstFile::info()` call is primarily for users that don't have access to a `M The downside is that `open()` requires a read to obtain the `SsTableHandle` even if the caller only wants to call `metadata()`, which doesn't need it. This is a fine tradeoff. -`index()` calls `SsTableFormat::read_index()`, which reads `info.index_offset..info.index_offset + info.index_len`, decompresses, and returns an `SsTableIndexOwned`. The method materializes `Vec<(u64, Bytes)>` from the FlatBuffer `BlockMeta` entries (each has `offset()` and `first_key()`). Caching uses `DbCache::get_index` / `insert` keyed by `(sst_id, index_offset)`, matching the existing pattern in `TableStore::read_index()`. +`index()` calls `SsTableFormat::read_index()`, which reads `info.index_offset..info.index_offset + info.index_len`, decompresses, and returns an `SsTableIndexOwned`. The returned `SstIndex` retains the cached `Arc` and reads FlatBuffer `BlockMeta` entries without copying their keys. Caching is handled by `TableStore::read_index()`. The existing `SstFileMetadata` struct in `tablestore.rs` (currently `pub(crate)`) is made `pub`. @@ -253,7 +258,7 @@ For cardinality: open each covering SST with `SstReader` (via `view.sst`) and ca #### Refined estimate — block-level for boundary SSTs -Most `SsTableView`s returned by `tables_covering_range()` are fully contained within the query range — their stats apply directly. Only the first and last view in each sorted run partially overlap. For these two boundary SSTs, call `sst_file.index()` to get the index `[(offset, first_key), ...]`. Binary search for the range start key in the first boundary SST to find where the range begins; binary search for the range end key in the last boundary SST to find where it ends. Note that an `SsTableView` may have a `visible_range()` projection that further restricts the effective key range — the query range should be intersected with the view's visible range before performing the binary search. +Most `SsTableView`s returned by `tables_covering_range()` are fully contained within the query range — their stats apply directly. Only the first and last view in each sorted run partially overlap. For these two boundary SSTs, call `sst_file.index()` to get an `SstIndex`. Use `partition_point()` to find the range start in the first boundary SST and the range end in the last boundary SST. Note that an `SsTableView` may have a `visible_range()` projection that further restricts the effective key range — the query range should be intersected with the view's visible range before searching the index. These offsets are compressed/stored sizes since the block index tracks on-disk offsets. @@ -398,7 +403,7 @@ SST stats block: - Record counting requires both stats (for `block_stats`) and index (for binary search on keys). Approximate count: stats + index reads. Exact count: + at most 2 data block reads per boundary SST. `SstFile::index()`: -- One index block read per SST, cacheable via the block cache. No changes to `BlockMeta` format. +- One index block read per SST, cacheable via the block cache. The returned `SstIndex` retains the cached data and exposes borrowed first keys without copying them. No changes to `BlockMeta` format. Memtable metrics via `Db::metrics()`: - No I/O. Reads atomic counters. @@ -435,7 +440,7 @@ Unit tests: - `SstReader::open()`: loading SST footer and constructing `SstFile` - `SstReader::open_with_handle()`: constructing `SstFile` from an existing `SsTableHandle` - `SstFile::stats()`: correct reading and population of `SstStats` from the stats block -- `SstFile::index()`: returns correct `(offset, first_key)` pairs matching the SST's block index +- `SstFile::index()`: returns an `SstIndex` whose accessors expose the correct `(offset, first_key)` pairs and partition points - `block_stats` vector: parallel to index, builder correctly tracks per-block put/delete/merge counts - Backward compatibility: old SSTs without stats return `None` - `Db::manifest()`: returns current manifest state with L0 and sorted runs @@ -502,4 +507,5 @@ Another alternative not explored is sample-based estimation: sample N random blo - **2026-02-16**: Stats fields moved into `SsTableInfo` (in `sst.fbs`) instead of a separate footer block. Removed `SstStats` struct — `SstFile::info()` returns `SsTableInfo` directly. `SstFile` now holds `SsTableHandle` + `Arc`. Added `object_store_cache_options` parameter to `SstReader::new()`. (PR #1220 review feedback from @criccomini). - **2026-02-19**: Reverted to separate stats block approach. Stats fields moved back out of `SsTableInfo` into a dedicated stats block within the SST file, referenced by `stats_offset`/`stats_len` in `SsTableInfo`. Reintroduced `SstStats` struct and `SstFile::stats()` method. This keeps `SsTableInfo` (and the manifest) lean — 16 bytes per SST vs 40 bytes — which matters for large DBs. Added `SstReader::open_with_handle()` for zero-I/O construction from an existing `SsTableHandle`. (PR #1220 review feedback from @rodesai and @criccomini). - **2026-02-25**: Added per-block record counts as `block_stats: [BlockStats]` in `SstStats` (stats block). `BlockStats` contains `num_puts`/`num_deletes`/`num_merges`, mirroring the SST-level aggregate fields. `BlockStats` uses a FlatBuffers `table` for future extensibility. -- **2026-03-19**: Updated RFC to reflect `SsTableView` indirection introduced in [#1362](https://github.com/slatedb/slatedb/pull/1362). `ManifestCore` and `SortedRun` now contain `SsTableView` references instead of raw `SsTableHandle`s. Updated `tables_covering_range()` return type to `VecDeque<&SsTableView>`, replaced `SsTableHandle::estimate_size()`/`visible_range()` references with `SsTableView` equivalents, and added "Interaction with `SsTableView`" section noting that `SstReader::open_with_handle()` accepts `SsTableHandle` via `view.sst`. Refined estimate section updated to account for view-level `visible_range` projections on boundary SSTs. \ No newline at end of file +- **2026-03-19**: Updated RFC to reflect `SsTableView` indirection introduced in [#1362](https://github.com/slatedb/slatedb/pull/1362). `ManifestCore` and `SortedRun` now contain `SsTableView` references instead of raw `SsTableHandle`s. Updated `tables_covering_range()` return type to `VecDeque<&SsTableView>`, replaced `SsTableHandle::estimate_size()`/`visible_range()` references with `SsTableView` equivalents, and added "Interaction with `SsTableView`" section noting that `SstReader::open_with_handle()` accepts `SsTableHandle` via `view.sst`. Refined estimate section updated to account for view-level `visible_range` projections on boundary SSTs. +- **2026-08-11**: Changed the return type of `SstFile::index` from `Vec<(u64, Bytes)>` to `SstIndex`. `SstIndex` is an opaque view into the underlying FlatBuffer via `SsTableIndexOwned`. `SstIndex` exposes block offsets and first keys via an `ExactSizeIterator` and lets the user search for a block via a `partition_point` method that uses the SlateDB internal binary search for finding a block offset. This is a breaking change as the return type of `SstFile::index` changes. diff --git a/rfcs/0024-segment-oriented-compaction.md b/rfcs/0024-segment-oriented-compaction.md index 67fd2922be..4fe6afdb0e 100644 --- a/rfcs/0024-segment-oriented-compaction.md +++ b/rfcs/0024-segment-oriented-compaction.md @@ -336,10 +336,11 @@ The extractor must be configured when the database is first created, or never co **Projection.** For each segment in `segments`, apply the same view-intersection rules as for the unsegmented `l0` and `compacted` lists: drop SST views whose effective range lies fully outside the projection range, and tag boundary views with a `visible_range`. Segments whose views are all excluded are removed from `segments`. The `segment_extractor_name` field is preserved unchanged. After projection a segment's effective range may be narrower than `[prefix, prefix++)`; this is benign, as `visible_range` enforcement on each view governs read and write access. -**Union.** Union of N segmented manifests adds two preconditions on top of those in [RFC 0004](./0004-checkpoints.md#union): +**Union.** Union of N segmented manifests relaxes one precondition from [RFC 0004](./0004-checkpoints.md#union) and adds two: +- The non-overlapping key range precondition applies per segment, not to each source's manifest as a whole. A read routes to exactly one segment, and each segment's chain is built only from that prefix's entries, so sources that overlap across *different* segments never collide; only sources contributing to the same prefix must be disjoint. This is what makes a segmented database rescalable along a dimension the segment prefix does not lead with. Shards holding `data/{tenant}` and `idx/{tenant}` rows, split by tenant, each span both segments — their bounding ranges overlap while no single segment does, so they union in one operation instead of through staged re-slicing clones. - All sources must share the same `segment_extractor_name` exactly — every source `None`, or every source the same `Some(name)`. Mixed configurations are rejected. Although unioned ranges are disjoint, we don't know that unsegmented data from a no-extractor source will remain unsegmented after union: a key persisted in `core.tree` may match an extractor prefix carried over from another source, and a future read of that key would route through the extractor to a segment that does not contain it, dropping the value. A future extension can relax this once a per-key check confirms unsegmented data does not match any extractor prefix; for now we require exact agreement. -- The combined set of segment prefixes across all sources must form an antichain (no prefix is a proper prefix of another). This usually follows from the existing non-overlapping key range precondition, but is checked explicitly to defend against stale extractor-name matches. +- The combined set of segment prefixes across all sources must form an antichain (no prefix is a proper prefix of another). With the key range precondition now scoped per segment, this no longer follows from it at all, and it also defends against stale extractor-name matches. For each segment prefix in the inputs: diff --git a/rfcs/0030-pluggable-wal.md b/rfcs/0030-pluggable-wal.md index b18081daaa..b34cef000d 100644 --- a/rfcs/0030-pluggable-wal.md +++ b/rfcs/0030-pluggable-wal.md @@ -88,15 +88,15 @@ replays these WAL files into memtables, filtering out any rows with sequence num **Writes** -Once it's recovered persisted writes, the db hands the WAL (`WalBufferManager`) off to the -Batch Writer task. This task serializes all writes and buffers them in `WalBufferManager`, -which periodically flushes the writes to a new WAL file. `WalBufferManager` notifies blocked +Once it's recovered persisted writes, the db hands the WAL (`SlateDbWalWriter`) off to the +Batch Writer task. This task serializes all writes and buffers them in `SlateDbWalWriter`, +which periodically flushes the writes to a new WAL file. `SlateDbWalWriter` notifies blocked write tasks when writes are durably flushed. **Memtable/L0 Flushing** The Batch Writer task adds writes to the memtable once they've been buffered in -`WalBufferManager`. It "freezes" memtables once they cross the memtable size threshold and +`SlateDbWalWriter`. It "freezes" memtables once they cross the memtable size threshold and annotates the frozen memtable with a `replay_after_wal_id` which holds the ID of some WAL File whose writes are fully covered by the memtable (in the current implementation this is the last durably flushed WAL File). The frozen memtables are picked up by a separate Manifest Writer task, @@ -112,7 +112,7 @@ described above depending on what the user requested. **Checkpoints** -When `WalBufferManager` durably persists a WAL File, it notifies the db, which updates +When `SlateDbWalWriter` durably persists a WAL File, it notifies the db, which updates `last_seen_wal_id` in the manifest with the flushed WAL ID. `DbReader` uses this field to determine the range of WAL Files that should be read for a checkpoint. @@ -331,6 +331,18 @@ pub trait WalWriter: Send { /// future that receives the result of the flush once it completes. async fn flush(&mut self) -> Result; + /// Returns true if the WAL implementation wants to request that the current in-memory + /// writes be flushed to a new l0. WAL implementations can use this to (1) bound the range + /// of writes that need to be replayed when SlateDB restarts, and (2) push data to L0s earlier + /// so that it's available to readers, which poll the latest manifest. + /// + /// ## Arguments + /// - `replay_after_wal_id`: The WAL ID used as the replay point for the last memtable + /// that was flushed to L0 + fn should_flush_memtable(&self, _replay_after_wal_id: u64) -> bool { + false + } + /// Returns a `WalObserver` for reading [`WalStatus`] and subscribing to events. fn observer(&self) -> Box; @@ -347,14 +359,8 @@ pub struct WalRows { /// The rows read from the WAL File. All the rows with a given sequence number must be present /// in th same [`WalRows`]. pub rows: Vec, - /// The id of the last WAL File containing rows from `rows`. There may still be rows with higher - /// sequence numbers in the WAL File with this id. - pub last_wal_file_id: u64, - /// True when this batch is the last one in its WAL file. This is an - /// optimization, so its harmless to always set to false. Callers can already infer that a - /// file is fully applied when they see a batch from a later file, but this flag lets them - /// advance their WAL watermark over the current file without waiting for the next one. - pub last_in_file: bool, + /// The id of the last WAL File for which all rows have been consumed by the iterator. + pub last_consumed_wal_file_id: u64, } /// An iterator over rows in some range of the WAL @@ -383,11 +389,16 @@ pub trait WalReader { &self, wal_file_id_range: WalFileRange, ) -> Result, WalError>; + + /// Returns the ID of the last WAL file currently present after `replay_after_wal_id`, or + /// `replay_after_wal_id` if no later WAL file is present. Implementations may use + /// `replay_after_wal_id` as a known lower bound when locating the end of the WAL. + async fn last_wal_file_id(&self, replay_after_wal_id: u64) -> Result; } /// API for plugging into WAL GC #[async_trait] -pub trait WalGC { +pub trait WalGc { /// Hook for garbage collecting the WAL. Takes a list of ranges of WAL Files that are currently /// referenced by some active Manifest. The implementation may delete any WAL File that is not /// included in the ranges in this list. @@ -396,6 +407,65 @@ pub trait WalGC { referenced_ranges: Vec, ) -> Result<(), WalError>; } + +/// Administrative operations for a WAL implementation. +#[async_trait] +pub trait WalAdmin: Send + Sync + 'static { + /// Creates a garbage collector scoped to the WAL at `path`. + /// + /// ## Arguments + /// - `path`: The database path whose WAL should be garbage collected. + /// + /// ## Returns + /// A garbage collector that can remove unreferenced WAL files at `path`. + fn garbage_collector(&self, path: &Path) -> Box; + + /// Deletes the WAL at `path`. + /// + /// ## Arguments + /// - `path`: The database path whose WAL should be deleted. + /// + /// ## Returns + /// `Ok(())` after the WAL has been deleted, or a [`WalError`] if deletion fails. + async fn delete_wal(&self, path: &Path) -> Result<(), WalError>; + + /// Given a path and WAL ID range, returns true if the WAL at that path is empty within the + /// specified range. A WAL is empty if it holds no records. + /// + /// ## Arguments + /// - `path`: The database path containing the WAL. + /// - `replay_after_wal_id`: The exclusive lower bound of the WAL range to inspect. + /// - `wal_id_last_seen`: The inclusive upper bound of the WAL range to inspect. + /// + /// ## Returns + /// `Ok(true)` if the referenced WAL contains no records, `Ok(false)` if it contains records, + /// or a [`WalError`] if the WAL could not be inspected. + async fn is_empty( + &self, + path: &Path, + replay_after_wal_id: u64, + wal_id_last_seen: u64, + ) -> Result; + + /// Given a source path and manifest, copy the referenced WAL to a destination path and return + /// a replay range. This call must be idempotent (TODO: clarify) + /// + /// ## Arguments + /// - `from_path`: The db path that holds the source WAL range to be copied + /// - `from_manifest`: The source manifest that identifies the WAL to copy + /// - `to_path`: The db path of the clone that the WAL is being copied to. + /// + /// ## Returns + /// A (u64, u64) pair. The first item will be used as the replay start point (exclusive). The + /// second item should be the id of the last WAL file id in the copied WAL. + async fn clone_wal( + &self, + from_path: &Path, + from_manifest: VersionedManifest, + to_path: &Path, + ) -> Result<(u64, u64), WalError>; +} + ``` Users can configure a custom WAL for the writer and reader using the db Builder: @@ -563,6 +633,10 @@ Memtable and Db flushing stay the same. The Batch Writer task annotates each imm with a safe replay point using `WalStatus::last_flushed_wal_id`, and flushes the WAL using `WalWriter::flush`. +As part of this change we will move tracking of early memtable flush via +`max_wal_flushes_before_l0_flush` to the native WAL implementation in the implementation of +`WalWriter::should_flush_memtable`. + #### Garbage Collection `WalGcTask` lists manifests to determine the set of referenced WAL File IDs and then delegates @@ -574,39 +648,30 @@ then deletes them. `DbReaderBuilder` initializes `DbReader` with a `WalReader` that it uses to construct iterators for replaying the WAL when loading a checkpoint. -`DbReader` will now also continually stream WAL updates when configured to track the latest -writes. It does this by creating its `WalIterator` with an unbounded end range and blocking on -`next` from its background polling task. If the reader observes a `WalError::WalTruncated` then it +`DbReader` continually discovers WAL updates when configured to track the latest writes. It +resolves the current tail, creates a bounded `WalIterator`, drains it to completion, and then +refreshes the manifest before the next pass. If the reader observes a `WalError::WalTruncated`, it immediately refreshes the manifest. #### CDC -We'll deprecate/remove the current CDC API. Users can use the `WalReader`/`WalIterator` proposed -in this RFC. SlateDB's native `WalReader` will take a buffer size and a poll interval to use when -tailing the current WAL: +We'll deprecate/remove the file-listing CDC API. CDC users can use SlateDB's native +`SlateDbWalReader`, which implements the `WalReader`/`WalIterator` traits proposed in this RFC. -```rust -struct ObjectStoreWalReader { - // ... -} +CDC uses an iterator with an unbounded end range. A consumer keeps the last fully consumed WAL +file ID, creates one iterator over `(cursor + 1)..`, and persists each +`WalRows.last_consumed_wal_file_id`. At the current tail, `WalIterator::next` polls the manifest and +waits for the next WAL file rather than ending, so the caller does not alternate between tail +discovery and iterator creation. Empty fence WALs return an empty batch that still advances the +cursor. After a restart, the consumer creates a new iterator from the persisted cursor plus one. -impl ObjectStoreWalReader { - pub fn new>( - path: P, - object_store: Arc, - /// The number of WAL Files to prefetch and buffer when streaming the WAL - buffered_files: usize, - /// The interval at which the next WAL file will be polled when streaming the latest updates - poll_interval: Duration - ) { - todo!() - } -} +#### Clones -impl WalReader for ObjectStoreWalReader { - // ... -} -``` +Clone creation delegates to `WalAdmin::empty` to introspect source WALs to determine if they are +empty when validating that unions/projections don't require copying the WAL. + +Clone creation delegates to `WalAdmin::clone_wal` to copy the WAL from the source db to the clone +db. #### Error Handling @@ -714,8 +779,8 @@ implementation is correct. Some important test cases we'll cover (non-exhaustive - `WalWriter` emits events when rows are durably stored. - `WalIterator` always iterates over writes in sequence order - `WalIterator` always returns full write batches in `WalRows` -- `WalIterator` tracks the WAL file id in `WalRows` correctly (TODO: this probably needs some - test interfaces in the reader for listing/reading wal files) +- `WalIterator` tracks the last consumed WAL file id in `WalRows` correctly (TODO: this probably + needs some test interfaces in the reader for listing/reading wal files) **Performance** @@ -730,8 +795,8 @@ benchmark tools that instantiate `DbBench` with a db configured to use a custom ## Packaging -The WAL traits and conformance tests will reside in a new crate called `slatedb-wal`. The native -WAL implementation remains in `slatedb`. +The WAL traits and conformance tests will all reside in the `wal` module. The native +WAL implementation will move to a nested module `wal::slatedb`. ## Alternatives @@ -772,6 +837,15 @@ We could have `WalReader`/`WalIterator` iterate over WAL Files which in turn sup iteration (similar to the CDC `WalReader`). I don't really see the benefit of imposing the extra layering. It also forces implementations to map each write batch to a single WAL File. +**Put WAL Trait Definitions in Separate Create** +Initially this RFC proposed putting the WAL trait defs in a separate crate so that implementors +only need to import that crate (vs all of slatedb). We opted not to go this route for a few reasons: +- This requires moving core slatedb types out to either the new wal crate or to slatedb-common. In + particular, we'd need to move all the manifest definitions (`VersionedManifest`, `Manifest`) and + `RowEntry` as these are used by the writer init and reader/iterator, respectively. +- A separate crate isn't that useful. Implementors will almost always have to import slatedb + anyway to run end-to-end tests. And users of custom WALs would be importing slatedb anyway. + ## Open Questions - ~~This RFC proposes an API for streaming new writes via `WalReader`/`WalIterator`. Should this be diff --git a/rfcs/0033-query_tracing.md b/rfcs/0033-query_tracing.md new file mode 100644 index 0000000000..5003d7aceb --- /dev/null +++ b/rfcs/0033-query_tracing.md @@ -0,0 +1,310 @@ +# SlateDB Query Tracing + +Table of Contents: + + + +- [Summary](#summary) +- [Motivation](#motivation) +- [Goals](#goals) +- [Non-Goals](#non-goals) +- [Design](#design) + * [Overview](#overview) + * [Query ID in `ReadOptions` and `ScanOptions`](#query-id-in-readoptions-and-scanoptions) + * [Tracing spans](#tracing-spans) +- [Impact Analysis](#impact-analysis) + * [Core API & Query Semantics](#core-api--query-semantics) + * [Consistency, Isolation, and Multi-Versioning](#consistency-isolation-and-multi-versioning) + * [Time, Retention, and Derived State](#time-retention-and-derived-state) + * [Metadata, Coordination, and Lifecycles](#metadata-coordination-and-lifecycles) + * [Compaction](#compaction) + * [Storage Engine Internals](#storage-engine-internals) + * [Ecosystem & Operations](#ecosystem--operations) +- [Operations](#operations) + * [Performance & Cost](#performance--cost) + * [Observability](#observability) + * [Compatibility](#compatibility) +- [Testing](#testing) +- [Rollout](#rollout) +- [Alternatives](#alternatives) + * [Recording aggregations and spans](#recording-aggregations-and-spans) + * [Recording the aggregations in a tracing subscriber](#recording-the-aggregations-in-a-tracing-subscriber) +- [Open Questions](#open-questions) +- [References](#references) +- [Updates](#updates) + + + +Status: Draft + +Authors: + +* [Almog Gavra](https://github.com/agavra) +* [Bruno Cadonna](https://github.com/cadonna) + +## Summary + +This RFC proposes to instrument the read path of SlateDB with the `tracing` library. The read path is instrumented +per query, i.e., per call to a `get` or `scan` method. The instrumentation consists of a root span that +tracks a complete `get` or `scan` operation (including calls on the returned iterator) and child spans that track various +stages of the read path. For example, looking up an entry in the memtable produces a child span. Another example is +reading a filter of an SST. + +The instrumentation is disabled by default. It can be enabled per `get` or `scan` operation by setting the tracing +options in the options of the read operation, i.e., in `ReadOptions` and in `ScanOptions`. + +The generated spans contain fields with information about the span. Each span of a specific read +operation contains the trace ID set in the tracing options passed to the read operation. In addition, the spans contain +information specific to the span, such as the ID of the SST if the span traces processing related to an SST. + +We do not propose any tracing subscriber. The produced spans can be processed by an existing tracing subscriber. +For example, `tracing-chrome` can be used to visualize the spans. + +## Motivation + +SlateDB tracks read-path statistics (bloom filter hits, request counts) +via global `DbStats` counters backed by the `MetricsRecorder` system +(RFC-0021). These aggregate counters answer "how is the system doing?" +but not "why was *this* query slow?" or "how many SSTs did my point +lookup touch?" + +Users today cannot: + +- Determine whether a slow get was caused by bloom filter false + positives, cache misses, or scanning too many L0 SSTs +- Measure how much wall-clock time a scan spent reading blocks from + object storage vs. serving from cache +- Write tests that assert query execution characteristics (e.g. + "this get should hit the bloom filter and skip the SST") + +## Goals + +- Per-query instrumentation via `tracing` spans (e.g., filter evaluations, index reads, + block reads). +- Zero overhead when not opted in (single `Option` branch skip). +- No changes to `DbRead` trait signatures or public API beyond adding + a field to existing options structs. + +## Non-Goals + +- Replacing or duplicating the global `MetricsRecorder` system. Both `DbStats` (aggregate) and the tracing spans + report to distinct consumers independently. +- Write-path tracing (puts, deletes, flush). +- A new `tracing` subscriber/layer. Users should use an existing `tracing` subscriber/layer, such as `tracing-chrome`, + or implement their own subscriber/layer to process and visualize spans. + +## Design + +### Overview + +This proposal adds two concepts to the read path: + +1. Optional tracing options to `ReadOptions` and `ScanOptions`. The default of the tracing options + is `None`, i.e., tracing is disabled by default. +2. `tracing` spans that are conditionally created when the options passed to `get*()` and `scan*()` carry tracing + options that is not `None`. + +### Tracing options in `ReadOptions` and `ScanOptions` + +`ReadOptions` and `ScanOptions` are extended with optional tracing options. +If the tracing options are set, the read path is instrumented. Otherwise, +SlateDB does not create any instrumentation on the read path. + +```rust +pub struct TracingOptions { // new + pub trace_id: String, +} + +pub struct ReadOptions { + pub durability_filter: DurabilityLevel, + pub dirty: bool, + pub cache_blocks: bool, + pub filter_context: Option, + pub tracing_options: Option, // new +} + +pub struct ScanOptions { + pub durability_filter: DurabilityLevel, + pub dirty: bool, + pub read_ahead_bytes: usize, + pub cache_blocks: bool, + pub max_fetch_tasks: usize, + pub order: IterationOrder, + pub filter_context: Option, + pub tracing_options: Option, // new +} +``` + +Both get a `with_tracing_options(TracingOptions) -> Self` builder method. Default is +`None`. + +### Tracing spans + +| Span | Recorded fields | +|--------------------------------|-------------------------------------------------------------| +| `slatedb.read` | `trace_id` | +| `slatedb.read.memtable` | `trace_id` | +| `slatedb.read.read_filters` | `trace_id`, `sst_id`, `level`, `cached` | +| `slatedb.read.evaluate_filter` | `trace_id`, `sst_id`, `level`, `name`, `result` | +| `slatedb.read.read_index` | `trace_id`, `sst_id`, `level`, `cached` | +| `slatedb.read.read_blocks` | `trace_id`, `sst_id`, `level`, `cache_hits`, `cache_misses` | +| `slatedb.read.merge` | `trace_id`, `num_operands` | + +The read path spans are structured hierarchically. The root span for the read path is named `slatedb.read`. +All others are direct children of `slatedb.read`. All spans carry the trace ID +(`trace_id`) as field. The spans are all constructed at debug level. +Spans instrumented on a future are entered each time the future is polled by the runtime. + +The root span `slatedb.read` traces the entire read operation, which includes all stages of the read path +covered by the child spans and common operations over all sources needed for reading, such as setting up +iterators. For scans, the span also covers the read operations triggered by calls on the returned lazy iterator. + +Span `slatedb.read.memtable` traces lookups on the active memtable and the immutable memtables. + +Spans `slatedb.read.read_filters` and `slatedb.read.read_index` trace the reading of filters and reading of the index +of an SST, respectively. The spans carry the ID of the SST (`sst_id`) the filters and the index belong to, the +level on which the SST resides (`level=l0` or `level=sorted_run:{id}`), and field `cached` +that records if the filters or index were found in the cache (`cached=true`) or not (`cached=false`). If the cache +is disabled, `cached` will be `false`. + +The evaluation of a single filter is tracked by span `slatedb.read.evaluate_filter`. The span exposes fields for +the SST ID, the level of the SST, the name of the filter (`name`), and the result of the evaluation (`result`). +For the built-in bloom filter, the `name` field will contain `_bf`. + +Span `slatedb.read.read_blocks` traces the reading of data blocks of an SST. The fields of the span hold the SST ID, +the level of the SST, and how many cache hits and misses were encountered while reading the blocks. + +Processing of the merge operator is traced by span `slatedb.read.merge`. Merging is performed in batches. For each +batch a separate span is produced. Each span contains the number of merged operands as a field. + +## Impact Analysis + +SlateDB features and components that this RFC interacts with. Check all that apply. + +### Core API & Query Semantics + +- [x] Basic KV API (`get`/`put`/`delete`) +- [x] Range queries, iterators, seek semantics +- [ ] Range deletions +- [ ] Error model, API errors + +### Consistency, Isolation, and Multi-Versioning + +- [ ] Transactions +- [ ] Snapshots +- [ ] Sequence numbers + +### Time, Retention, and Derived State + +- [ ] Time to live (TTL) +- [ ] Compaction filters +- [x] Merge operator +- [ ] Change Data Capture (CDC) + +### Metadata, Coordination, and Lifecycles + +- [ ] Manifest format +- [ ] Checkpoints +- [ ] Clones +- [ ] Garbage collection +- [ ] Database splitting and merging +- [ ] Multi-writer + +### Compaction + +- [ ] Compaction state persistence +- [ ] Compaction filters +- [ ] Compaction strategies +- [ ] Distributed compaction +- [ ] Compactions format + +### Storage Engine Internals + +- [ ] Write-ahead log (WAL) +- [x] Block cache +- [x] Object store cache +- [x] Indexing (bloom filters, metadata) +- [ ] SST format or block format + +### Ecosystem & Operations + +- [ ] CLI tools +- [x] Language bindings (Go/Python/etc) +- [x] Observability (metrics/logging/tracing) + +## Operations + +### Performance & Cost + +The proposed instrumentation is disabled by default. Reads without tracing options should not change performance or cost. +When tracing options are set but no tracing subscriber/layer is configured, performance should not be significantly affected. +Set tracing options and a configured tracing subscriber/layer might negatively affect performance. The performance of +writes and compactions should not be affected at all. + +### Observability + +Observability is extended by a per-query instrumentation that traces the read path. The instrumentation only produces +traces at debug level and only if a tracing subscriber/layer is configured. + +### Compatibility + +Field `tracing_options` is added to the public API `ReadOptions` and `ScanOptions`. The field is also exposed in the bindings. +Since the default value of field `tracing_options` is `None`, read path tracing is disabled for existing queries. + +## Testing + +- Unit tests: + - For each kind of span +- Integration tests: + - With subscriber and different queries +- Performance tests: + - With and without tracing options, + - With and without subscriber + - At info and debug level + +## Rollout + +- Milestones / phases: + - Adding root span `slatedb.read`. + - Adding span `slatedb.read.memtable`. + - Adding span `slatedb.read.read_filters`. + - Adding span `slatedb.read.evaluate_filter`. + - Adding span `slatedb.read.read_index`. + - Adding span `slatedb.read.read_blocks`. + - Adding span `slatedb.read.merge`. + - Performance experiments. +- Docs updates: + - Documentation of spans + - Usage example + +## Alternatives + +### Recording aggregations and spans + +The idea was to pass a struct to `ReadOptions` and `ScanOptions` to record and aggregate various measurements, such as +the number of accesses to different sources (e.g., memtables and SSTs), cache misses, and cache hits, as well as instrumenting +spans for recording execution times. This was rejected because collecting aggregations overlapped with instrumenting +the code with spans. We decided to consolidate measurement collection on the read path and also wanted to reduce code +complexity. + +### Recording the aggregations in a tracing subscriber + +This approach consisted of instrumenting the read path and creating a tracing subscriber specifically for SlateDB that +also maintains aggregations. The tracing subscriber would process the instrumented spans and offer an API to read the +recorded data on the read path. This approach was rejected because of the complexity and maintenance burden. +There are already existing tracing subscribers, e.g., `tracing-chrome`, that can be used for analyzing an instrumented +read path. We decided to start with the instrumentation and postpone a dedicated tracing subscriber to the future if +required. + +## Open Questions + +- Should we add a field to the `read_filters` span that records how many filters are read from the SST? + +## References + +- `tracing` crate: https://crates.io/crates/tracing +- `tracing-chrome`: https://crates.io/crates/tracing-chrome +- https://github.com/slatedb/slatedb/issues/400 +- https://github.com/slatedb/slatedb/issues/797 + +## Updates diff --git a/slatedb-bencher/README.md b/slatedb-bencher/README.md index 7aca293c36..4c8ba6f97f 100644 --- a/slatedb-bencher/README.md +++ b/slatedb-bencher/README.md @@ -58,10 +58,10 @@ following environment variables before benchmarking: ## `benchmark-db.sh` -There is also a shell script which runs a series of benchmarks and then draws -the plots using `gnuplot`. Think of it as a template to start with to create -a set of benchmarks suitable for your task. The script should be run from -the repository root: +There is also a shell script which runs a series of benchmarks and records +the results. Think of it as a template to start with to create a set of +benchmarks suitable for your task. The script should be run from the repository +root: ```bash ./slatedb-bencher/benchmark-db.sh @@ -69,16 +69,32 @@ the repository root: The command above will produce results at `target/bencher/results` directory. The results include: -- `plots`: Plots for each benchmark - `dats`: Data files for each benchmark - `logs`: Log files for each benchmark -- `benchmark-data.json`: A JSON file containing all the benchmark results in [github-action-benchmark](https://github.com/benchmark-action/github-action-benchmark) format. -The script also has a `SLATEDB_BENCH_CLEAN` environment variable which can be set to `true` to clean up the test data in object storage after each benchmark. +### Plotting results with `gnuplot` + +The `.dat` files are whitespace-delimited, with columns for elapsed time, +puts per second, and gets per second. After installing `gnuplot`, you can render +a result file to a PNG with: + +```bash +gnuplot <<'EOF' +set terminal pngcairo size 1280,720 +set output "target/bencher/results/20_1.png" +set title "SlateDB benchmark: 20% puts, concurrency 1" +set xlabel "Elapsed time (seconds)" +set ylabel "Requests per second" +set key outside +plot "target/bencher/results/dats/20_1.dat" using 1:2 with lines title "puts/s", \ + "target/bencher/results/dats/20_1.dat" using 1:3 with lines title "gets/s" +EOF +``` -### `nightly.yaml` +Replace `20_1.dat` and the labels with the benchmark configuration you want to +plot. -`benchmark-db.sh` is also used in `.github/workflows/nightly.yaml` to benchmark the nightly build. The tests are run using [WarpBuild](https://warpbuild.com), and each run appends to mermaid `xyChart` files that are posted to the workflow's GitHub Actions job summary. +The script also has a `SLATEDB_BENCH_CLEAN` environment variable which can be set to `true` to clean up the test data in object storage after each benchmark. ## `compaction` Subcommand diff --git a/slatedb-bencher/benchmark-db.sh b/slatedb-bencher/benchmark-db.sh index 9ea2d3a7d0..f4d1978490 100755 --- a/slatedb-bencher/benchmark-db.sh +++ b/slatedb-bencher/benchmark-db.sh @@ -8,7 +8,6 @@ OUT="target/bencher/results" mkdir -p $OUT/dats mkdir -p $OUT/logs -mkdir -p $OUT/mermaid run_bench() { local put_percentage="$1" @@ -46,165 +45,6 @@ generate_dat() { grep "stats dump" "$input_file" | sed -E 's/.*elapsed ([0-9.]+).*put\/s: ([0-9.]+).*get\/s: ([0-9.]+).*/\1 \2 \3/' > "$output_file" } -generate_mermaid () { - local dat_file="$1" - local mermaid_file="$2" - - # Create mermaid directory if it doesn't exist - mkdir -p "$(dirname "$mermaid_file")" - - # Get the last line from dat file (most recent benchmark result) - if [ ! -f "$dat_file" ] || [ ! -s "$dat_file" ]; then - echo "Warning: dat file $dat_file does not exist or is empty" - return 1 - fi - - local last_line=$(tail -n 1 "$dat_file") - local put_value=$(echo "$last_line" | awk '{print $2}') - local get_value=$(echo "$last_line" | awk '{print $3}') - - # Get git commit hash (first 7 characters) - local git_hash=$(git rev-parse --short=7 HEAD 2>/dev/null || echo "unknown") - - # Get current date in YYYY-MM-DD format - local current_date=$(date +"%Y-%m-%d") - - # Create x-axis entry - local x_entry="$current_date ($git_hash)" - - # Extract put_percentage and concurrency from mermaid filename - local filename=$(basename "$mermaid_file" .mermaid) - local put_percentage=$(echo "$filename" | cut -d'_' -f1) - local concurrency=$(echo "$filename" | cut -d'_' -f2) - - # Calculate max value for y-axis scaling - local max_value=$(echo "$put_value $get_value" | tr ' ' '\n' | sort -nr | head -n1) - local y_max=$(echo "$max_value * 1.2" | bc -l | cut -d'.' -f1) - - if [ ! -f "$mermaid_file" ]; then - # Create new mermaid file - cat > "$mermaid_file" << EOF ---- -config: - xyChart: - chartOrientation: horizontal - height: 768 - width: 1024 - themeVariables: - xyChart: - plotColorPalette: '#1e81b0, #e28743' ---- -xychart-beta - title "SlateDB [puts=${put_percentage}%, threads=${concurrency}, 🔵=puts, 🟠=get]" - x-axis ["$x_entry"] - y-axis "requests/s" 0 --> $y_max - line [$put_value] - line [$get_value] -EOF - else - # Update existing mermaid file - local temp_file=$(mktemp) - - # Read current content (match only Mermaid series lines, not YAML like plotColorPalette) - local title_line=$(grep -E "^[[:space:]]*title[[:space:]]" "$mermaid_file" | sed 's/^[[:space:]]*//') - local x_axis_line=$(grep -E "^[[:space:]]*x-axis[[:space:]]*\\[" "$mermaid_file") - local put_line=$(grep -E "^[[:space:]]*line[[:space:]]*\\[" "$mermaid_file" | head -n1) - local get_line=$(grep -E "^[[:space:]]*line[[:space:]]*\\[" "$mermaid_file" | tail -n1) - - # Extract current values - local current_x_values=$(echo "$x_axis_line" | sed 's/.*\[//;s/\].*//' | tr ',' '\n' | sed 's/^[[:space:]]*"//;s/"[[:space:]]*$//') - local current_put_values=$(echo "$put_line" | sed 's/.*\[//;s/\].*//') - local current_get_values=$(echo "$get_line" | sed 's/.*\[//;s/\].*//') - - # Convert to arrays - local x_array=() - local put_array=() - local get_array=() - - # Parse existing x-axis values - while IFS= read -r line; do - if [ -n "$line" ]; then - x_array+=("$line") - fi - done <<< "$current_x_values" - - # Parse existing put values - IFS=',' read -ra put_array <<< "$current_put_values" - - # Parse existing get values - IFS=',' read -ra get_array <<< "$current_get_values" - - # Prepend new values (newest first) - x_array=("$x_entry" "${x_array[@]}") - put_array=("$put_value" "${put_array[@]}") - get_array=("$get_value" "${get_array[@]}") - - # Keep only first 30 values (newest-first) if we have more - if [ ${#x_array[@]} -gt 30 ]; then - x_array=("${x_array[@]:0:30}") - put_array=("${put_array[@]:0:30}") - get_array=("${get_array[@]:0:30}") - fi - - # Build new x-axis string - local new_x_axis="x-axis [" - for i in "${!x_array[@]}"; do - if [ $i -gt 0 ]; then - new_x_axis="$new_x_axis, " - fi - new_x_axis="$new_x_axis\"${x_array[i]}\"" - done - new_x_axis="$new_x_axis]" - - # Build new put line string - local new_put_line="line [" - for i in "${!put_array[@]}"; do - if [ $i -gt 0 ]; then - new_put_line="$new_put_line, " - fi - new_put_line="$new_put_line${put_array[i]}" - done - new_put_line="$new_put_line]" - - # Build new get line string - local new_get_line="line [" - for i in "${!get_array[@]}"; do - if [ $i -gt 0 ]; then - new_get_line="$new_get_line, " - fi - new_get_line="$new_get_line${get_array[i]}" - done - new_get_line="$new_get_line]" - - # Calculate max value for y-axis scaling from all values - local all_values="${put_array[*]} ${get_array[*]}" - local max_value=$(echo "$all_values" | tr ' ' '\n' | sort -nr | head -n1) - local y_max=$(echo "$max_value * 1.2" | bc -l | cut -d'.' -f1) - - # Write updated mermaid file - cat > "$mermaid_file" << EOF ---- -config: - xyChart: - chartOrientation: horizontal - height: 768 - width: 1024 - themeVariables: - xyChart: - plotColorPalette: '#1e81b0, #e28743' ---- -xychart-beta - $title_line - $new_x_axis - y-axis "requests/s" 0 --> $y_max - $new_put_line - $new_get_line -EOF - fi - - echo "Generated/updated mermaid chart: $mermaid_file" -} - # Set CLOUD_PROVIDER to local if not already set export CLOUD_PROVIDER=${CLOUD_PROVIDER:-local} echo "Using cloud provider: $CLOUD_PROVIDER" @@ -220,11 +60,9 @@ for put_percentage in 20 40 60 80 100; do for concurrency in 1 32; do log_file="$OUT/logs/${put_percentage}_${concurrency}.log" dat_file="$OUT/dats/${put_percentage}_${concurrency}.dat" - mermaid_file="$OUT/mermaid/${put_percentage}_${concurrency}.mermaid" num_keys=$((put_percentage * 1000)) run_bench "$put_percentage" "$concurrency" "$num_keys" "$log_file" generate_dat "$log_file" "$dat_file" - generate_mermaid "$dat_file" "$mermaid_file" done done diff --git a/slatedb-bencher/benchmark-transaction.sh b/slatedb-bencher/benchmark-transaction.sh index 3f0e25f71b..55d6cc6311 100755 --- a/slatedb-bencher/benchmark-transaction.sh +++ b/slatedb-bencher/benchmark-transaction.sh @@ -7,7 +7,6 @@ OUT="target/bencher/transaction-results" mkdir -p "$OUT/logs" mkdir -p "$OUT/dats" -mkdir -p "$OUT/mermaid" # Define DB path once for both bencher and cleanup DB_PATH_NAME="slatedb-txn-bencher" @@ -72,195 +71,6 @@ generate_dat() { fi } -generate_mermaid() { - local dat_file="$1" - local mermaid_file="$2" - - # Create mermaid directory if it doesn't exist - mkdir -p "$(dirname "$mermaid_file")" - - # Get the last line from dat file (most recent benchmark result) - if [ ! -f "$dat_file" ] || [ ! -s "$dat_file" ]; then - echo "Warning: dat file $dat_file does not exist or is empty" - return 1 - fi - - local last_line=$(tail -n 1 "$dat_file") - local commit_value=$(echo "$last_line" | awk '{print $2}') - local abort_value=$(echo "$last_line" | awk '{print $3}') - local conflict_value=$(echo "$last_line" | awk '{print $4}') - local ops_value=$(echo "$last_line" | awk '{print $5}') - - # Get git commit hash (first 7 characters) - local git_hash=$(git rev-parse --short=7 HEAD 2>/dev/null || echo "unknown") - - # Get current date in YYYY-MM-DD format - local current_date=$(date +"%Y-%m-%d") - - # Create x-axis entry - local x_entry="$current_date ($git_hash)" - - # Extract test parameters from mermaid filename - # Format: isolation_concurrency_txnsize_mode.mermaid (e.g., snapshot_4_10_txn.mermaid) - local filename=$(basename "$mermaid_file" .mermaid) - local isolation=$(echo "$filename" | cut -d'_' -f1) - local concurrency=$(echo "$filename" | cut -d'_' -f2) - local txn_size=$(echo "$filename" | cut -d'_' -f3) - local mode=$(echo "$filename" | cut -d'_' -f4) - - local title_mode="Transaction" - if [ "$mode" = "batch" ]; then - title_mode="WriteBatch" - fi - - # Calculate max value for y-axis scaling (consider all 4 metrics) - local max_value=$(printf "%s\n" "$commit_value" "$abort_value" "$conflict_value" "$ops_value" | sort -nr | head -n1) - local y_max=$(awk -v m="$max_value" 'BEGIN{printf "%d", (m*1.2)}') - - if [ ! -f "$mermaid_file" ]; then - # Create new mermaid file - cat > "$mermaid_file" << EOF ---- -config: - xyChart: - chartOrientation: horizontal - height: 768 - width: 1024 - themeVariables: - xyChart: - plotColorPalette: '#2ecc71, #e74c3c, #f39c12, #3498db' ---- -xychart-beta - title "SlateDB Txn [${isolation}, threads=${concurrency}, txn_size=${txn_size}, ${title_mode}] 🟢=commit 🔴=abort 🟠=conflict 🔵=ops" - x-axis ["$x_entry"] - y-axis "requests/s" 0 --> $y_max - line [$commit_value] - line [$abort_value] - line [$conflict_value] - line [$ops_value] -EOF - else - # Update existing mermaid file - # Read current content (match only Mermaid series lines, not YAML like plotColorPalette) - local title_line=$(grep -E "^[[:space:]]*title[[:space:]]" "$mermaid_file" | sed 's/^[[:space:]]*//') - local x_axis_line=$(grep -E "^[[:space:]]*x-axis[[:space:]]*\\[" "$mermaid_file") - local commit_line=$(grep -E "^[[:space:]]*line[[:space:]]*\\[" "$mermaid_file" | sed -n '1p') - local abort_line=$(grep -E "^[[:space:]]*line[[:space:]]*\\[" "$mermaid_file" | sed -n '2p') - local conflict_line=$(grep -E "^[[:space:]]*line[[:space:]]*\\[" "$mermaid_file" | sed -n '3p') - local ops_line=$(grep -E "^[[:space:]]*line[[:space:]]*\\[" "$mermaid_file" | sed -n '4p') - - # Extract current values - local current_x_values=$(echo "$x_axis_line" | sed 's/.*\[//;s/\].*//' | tr ',' '\n' | sed 's/^[[:space:]]*"//;s/"[[:space:]]*$//') - local current_commit_values=$(echo "$commit_line" | sed 's/.*\[//;s/\].*//') - local current_abort_values=$(echo "$abort_line" | sed 's/.*\[//;s/\].*//') - local current_conflict_values=$(echo "$conflict_line" | sed 's/.*\[//;s/\].*//') - local current_ops_values=$(echo "$ops_line" | sed 's/.*\[//;s/\].*//') - - # Convert to arrays - local x_array=() - local commit_array=() - local abort_array=() - local conflict_array=() - local ops_array=() - - # Parse existing x-axis values - while IFS= read -r line; do - if [ -n "$line" ]; then - x_array+=("$line") - fi - done <<< "$current_x_values" - - # Parse existing values - IFS=',' read -ra commit_array <<< "$current_commit_values" - IFS=',' read -ra abort_array <<< "$current_abort_values" - IFS=',' read -ra conflict_array <<< "$current_conflict_values" - IFS=',' read -ra ops_array <<< "$current_ops_values" - - # Trim whitespace from array elements (comma-split can leave leading spaces) - # Use separate loops to handle potential length mismatches safely - for i in "${!commit_array[@]}"; do commit_array[i]="${commit_array[i]//[[:space:]]/}"; done - for i in "${!abort_array[@]}"; do abort_array[i]="${abort_array[i]//[[:space:]]/}"; done - for i in "${!conflict_array[@]}"; do conflict_array[i]="${conflict_array[i]//[[:space:]]/}"; done - for i in "${!ops_array[@]}"; do ops_array[i]="${ops_array[i]//[[:space:]]/}"; done - - # Prepend new values (newest first) - x_array=("$x_entry" "${x_array[@]}") - commit_array=("$commit_value" "${commit_array[@]}") - abort_array=("$abort_value" "${abort_array[@]}") - conflict_array=("$conflict_value" "${conflict_array[@]}") - ops_array=("$ops_value" "${ops_array[@]}") - - # Keep only first 30 values (newest-first) if we have more - if [ ${#x_array[@]} -gt 30 ]; then - x_array=("${x_array[@]:0:30}") - commit_array=("${commit_array[@]:0:30}") - abort_array=("${abort_array[@]:0:30}") - conflict_array=("${conflict_array[@]:0:30}") - ops_array=("${ops_array[@]:0:30}") - fi - - # Build new x-axis string - local new_x_axis="x-axis [" - for i in "${!x_array[@]}"; do - if [ $i -gt 0 ]; then - new_x_axis="$new_x_axis, " - fi - new_x_axis="$new_x_axis\"${x_array[i]}\"" - done - new_x_axis="$new_x_axis]" - - # Build new line strings - local new_commit_line="line [" - local new_abort_line="line [" - local new_conflict_line="line [" - local new_ops_line="line [" - for i in "${!commit_array[@]}"; do - if [ $i -gt 0 ]; then - new_commit_line="$new_commit_line, " - new_abort_line="$new_abort_line, " - new_conflict_line="$new_conflict_line, " - new_ops_line="$new_ops_line, " - fi - new_commit_line="$new_commit_line${commit_array[i]}" - new_abort_line="$new_abort_line${abort_array[i]}" - new_conflict_line="$new_conflict_line${conflict_array[i]}" - new_ops_line="$new_ops_line${ops_array[i]}" - done - new_commit_line="$new_commit_line]" - new_abort_line="$new_abort_line]" - new_conflict_line="$new_conflict_line]" - new_ops_line="$new_ops_line]" - - # Calculate max value for y-axis scaling from all values (all 4 metrics) - local max_value=$(printf "%s\n" "${commit_array[@]}" "${abort_array[@]}" "${conflict_array[@]}" "${ops_array[@]}" | sort -nr | head -n1) - local y_max=$(awk -v m="$max_value" 'BEGIN{printf "%d", (m*1.2)}') - - # Write updated mermaid file - cat > "$mermaid_file" << EOF ---- -config: - xyChart: - chartOrientation: horizontal - height: 768 - width: 1024 - themeVariables: - xyChart: - plotColorPalette: '#2ecc71, #e74c3c, #f39c12, #3498db' ---- -xychart-beta - $title_line - $new_x_axis - y-axis "requests/s" 0 --> $y_max - $new_commit_line - $new_abort_line - $new_conflict_line - $new_ops_line -EOF - fi - - echo "Generated/updated mermaid chart: $mermaid_file" -} - # Set CLOUD_PROVIDER to local if not already set export CLOUD_PROVIDER=${CLOUD_PROVIDER:-local} echo "Using cloud provider: $CLOUD_PROVIDER" @@ -283,41 +93,34 @@ echo "" echo "Test 1: Low concurrency with Transactions (Snapshot)" run_txn_bench "snapshot" 4 10 false "$OUT/logs/snapshot_4_10_txn.log" generate_dat "$OUT/logs/snapshot_4_10_txn.log" "$OUT/dats/snapshot_4_10_txn.dat" -generate_mermaid "$OUT/dats/snapshot_4_10_txn.dat" "$OUT/mermaid/snapshot_4_10_txn.mermaid" # Test 2: Low concurrency, Snapshot isolation, WriteBatch echo "Test 2: Low concurrency with WriteBatch" run_txn_bench "snapshot" 4 10 true "$OUT/logs/snapshot_4_10_batch.log" generate_dat "$OUT/logs/snapshot_4_10_batch.log" "$OUT/dats/snapshot_4_10_batch.dat" -generate_mermaid "$OUT/dats/snapshot_4_10_batch.dat" "$OUT/mermaid/snapshot_4_10_batch.mermaid" # Test 3: High concurrency, Snapshot isolation, Transaction echo "Test 3: High concurrency with Transactions (Snapshot)" run_txn_bench "snapshot" 16 10 false "$OUT/logs/snapshot_16_10_txn.log" generate_dat "$OUT/logs/snapshot_16_10_txn.log" "$OUT/dats/snapshot_16_10_txn.dat" -generate_mermaid "$OUT/dats/snapshot_16_10_txn.dat" "$OUT/mermaid/snapshot_16_10_txn.mermaid" # Test 4: High concurrency, Snapshot isolation, WriteBatch echo "Test 4: High concurrency with WriteBatch" run_txn_bench "snapshot" 16 10 true "$OUT/logs/snapshot_16_10_batch.log" generate_dat "$OUT/logs/snapshot_16_10_batch.log" "$OUT/dats/snapshot_16_10_batch.dat" -generate_mermaid "$OUT/dats/snapshot_16_10_batch.dat" "$OUT/mermaid/snapshot_16_10_batch.mermaid" # Test 5: High concurrency, SerializableSnapshot isolation echo "Test 5: High concurrency with SerializableSnapshot" run_txn_bench "serializable" 16 10 false "$OUT/logs/serializable_16_10_txn.log" generate_dat "$OUT/logs/serializable_16_10_txn.log" "$OUT/dats/serializable_16_10_txn.dat" -generate_mermaid "$OUT/dats/serializable_16_10_txn.dat" "$OUT/mermaid/serializable_16_10_txn.mermaid" # Test 6: Large transactions echo "Test 6: Large transactions (50 ops)" run_txn_bench "snapshot" 8 50 false "$OUT/logs/snapshot_8_50_txn.log" generate_dat "$OUT/logs/snapshot_8_50_txn.log" "$OUT/dats/snapshot_8_50_txn.dat" -generate_mermaid "$OUT/dats/snapshot_8_50_txn.dat" "$OUT/mermaid/snapshot_8_50_txn.mermaid" echo "" echo "=== Benchmark Complete ===" echo "Results saved to:" echo " Logs: $OUT/logs/" echo " Data: $OUT/dats/" -echo " Charts: $OUT/mermaid/" diff --git a/slatedb-bencher/src/db.rs b/slatedb-bencher/src/db.rs index 9b521bc074..46717a1bb0 100644 --- a/slatedb-bencher/src/db.rs +++ b/slatedb-bencher/src/db.rs @@ -87,7 +87,7 @@ impl RandomKeyGenerator { pub fn new(key_bytes: usize) -> Self { Self { key_len_bytes: key_bytes, - rng: rand_xorshift::XorShiftRng::from_os_rng(), + rng: XorShiftRng::from_os_rng(), used_keys: Vec::new(), } } @@ -134,7 +134,7 @@ impl FixedSetKeyGenerator { } Self { keys, - rng: rand_xorshift::XorShiftRng::from_os_rng(), + rng: XorShiftRng::from_os_rng(), used_keys: Vec::new(), } } @@ -162,6 +162,7 @@ pub struct DbBench { key_gen_supplier: Box Box>, val_len: usize, write_options: WriteOptions, + await_durable: bool, concurrency: u32, num_rows: Option, duration: Option, @@ -176,6 +177,7 @@ impl DbBench { key_gen_supplier: Box Box>, val_len: usize, write_options: WriteOptions, + await_durable: bool, concurrency: u32, num_rows: Option, duration: Option, @@ -187,6 +189,7 @@ impl DbBench { key_gen_supplier, val_len, write_options, + await_durable, concurrency, num_rows, duration, @@ -209,6 +212,7 @@ impl DbBench { (*self.key_gen_supplier)(), self.val_len, self.write_options.clone(), + self.await_durable, self.num_rows, self.duration, self.put_percentage, @@ -229,6 +233,7 @@ struct Task { key_generator: Box, val_len: usize, write_options: WriteOptions, + await_durable: bool, num_keys: Option, duration: Option, put_percentage: u32, @@ -243,6 +248,7 @@ impl Task { key_generator: Box, val_len: usize, write_options: WriteOptions, + await_durable: bool, num_keys: Option, duration: Option, put_percentage: u32, @@ -254,6 +260,7 @@ impl Task { key_generator, val_len, write_options, + await_durable, num_keys, duration, put_percentage, @@ -268,7 +275,7 @@ impl Task { /// This method runs a loop, generating a key (and value if needed), and /// then either puts the key/value pair or gets the key. async fn run(&mut self) { - let mut random = rand_xorshift::XorShiftRng::from_os_rng(); + let mut random = XorShiftRng::from_os_rng(); let mut puts = 0u64; let mut puts_bytes = 0u64; let mut gets = 0u64; @@ -283,12 +290,17 @@ impl Task { let key = self.key_generator.next_key(); let mut value = vec![0; self.val_len]; random.fill_bytes(value.as_mut_slice()); - match self + let result = self .db .put_with_options(key, value, &PutOptions::default(), &self.write_options) - .await - { - Ok(_) => { + .await; + let result = match result { + Ok(handle) if self.await_durable => handle.await_durable().await, + Ok(_) => Ok(()), + Err(error) => Err(error), + }; + match result { + Ok(()) => { puts += 1; puts_bytes += self.val_len as u64; } diff --git a/slatedb-bencher/src/main.rs b/slatedb-bencher/src/main.rs index 5101648cc4..22d085ad9b 100644 --- a/slatedb-bencher/src/main.rs +++ b/slatedb-bencher/src/main.rs @@ -83,10 +83,7 @@ async fn exec_benchmark_db(path: Path, object_store: Arc, args: if args.no_compactor { config.compactor_options = None; } - let write_options = WriteOptions { - await_durable: args.await_durable, - ..Default::default() - }; + let write_options = WriteOptions::default(); let mut builder = Db::builder(path.clone(), object_store.clone()).with_settings(config); @@ -99,6 +96,7 @@ async fn exec_benchmark_db(path: Path, object_store: Arc, args: args.key_gen_supplier(), args.val_len, write_options, + args.await_durable, args.concurrency, args.num_rows, args.duration.map(|d| Duration::from_secs(d as u64)), @@ -156,10 +154,7 @@ async fn exec_benchmark_transaction( args: BenchmarkTransactionArgs, ) { let (config, memory_cache) = args.db_args.config().unwrap(); - let write_options = WriteOptions { - await_durable: args.await_durable, - ..Default::default() - }; + let write_options = WriteOptions::default(); let mut builder = Db::builder(path.clone(), object_store.clone()).with_settings(config); @@ -173,6 +168,7 @@ async fn exec_benchmark_transaction( args.key_gen_supplier(), args.val_len, write_options, + args.await_durable, args.concurrency, args.duration.map(|d| Duration::from_secs(d as u64)), args.transaction_size, diff --git a/slatedb-bencher/src/transactions.rs b/slatedb-bencher/src/transactions.rs index 97fa0b7b78..d29c7d9a76 100644 --- a/slatedb-bencher/src/transactions.rs +++ b/slatedb-bencher/src/transactions.rs @@ -48,6 +48,7 @@ pub struct TransactionBench { key_gen_supplier: Box Box>, val_len: usize, write_options: WriteOptions, + await_durable: bool, concurrency: u32, duration: Option, transaction_size: u32, @@ -63,6 +64,7 @@ impl TransactionBench { key_gen_supplier: Box Box>, val_len: usize, write_options: WriteOptions, + await_durable: bool, concurrency: u32, duration: Option, transaction_size: u32, @@ -75,6 +77,7 @@ impl TransactionBench { key_gen_supplier, val_len, write_options, + await_durable, concurrency, duration, transaction_size, @@ -98,6 +101,7 @@ impl TransactionBench { (*self.key_gen_supplier)(), self.val_len, self.write_options.clone(), + self.await_durable, self.duration, self.transaction_size, self.abort_percentage, @@ -119,6 +123,7 @@ struct TransactionTask { key_generator: Box, val_len: usize, write_options: WriteOptions, + await_durable: bool, duration: Option, transaction_size: u32, abort_percentage: u32, @@ -134,6 +139,7 @@ impl TransactionTask { key_generator: Box, val_len: usize, write_options: WriteOptions, + await_durable: bool, duration: Option, transaction_size: u32, abort_percentage: u32, @@ -146,6 +152,7 @@ impl TransactionTask { key_generator, val_len, write_options, + await_durable, duration, transaction_size, abort_percentage, @@ -160,7 +167,7 @@ impl TransactionTask { /// /// This method runs a loop, executing transactions with multiple operations. async fn run(&mut self) { - let mut random = rand_xorshift::XorShiftRng::from_os_rng(); + let mut random = XorShiftRng::from_os_rng(); let mut commits = 0u64; let mut aborts = 0u64; let mut conflicts = 0u64; @@ -231,8 +238,14 @@ impl TransactionTask { batch.put(key, value); } - match self.db.write_with_options(batch, &self.write_options).await { - Ok(_) => Ok(ops as u64), + let result = self.db.write_with_options(batch, &self.write_options).await; + let result = match result { + Ok(handle) if self.await_durable => handle.await_durable().await, + Ok(_) => Ok(()), + Err(error) => Err(error), + }; + match result { + Ok(()) => Ok(ops as u64), Err(e) => { warn!("write batch failed [error={}]", e); Err(e) @@ -271,8 +284,14 @@ impl TransactionTask { return TransactionResult::Aborted; } - match txn.commit_with_options(&self.write_options).await { - Ok(_) => TransactionResult::Committed(ops as u64), + let result = txn.commit_with_options(&self.write_options).await; + let result = match result { + Ok(Some(handle)) if self.await_durable => handle.await_durable().await, + Ok(_) => Ok(()), + Err(error) => Err(error), + }; + match result { + Ok(()) => TransactionResult::Committed(ops as u64), Err(e) => { warn!("transaction commit failed (conflict) [error={}]", e); TransactionResult::Conflict diff --git a/slatedb-cli/src/args.rs b/slatedb-cli/src/args.rs index 838a7fd35b..3603121ce8 100644 --- a/slatedb-cli/src/args.rs +++ b/slatedb-cli/src/args.rs @@ -128,6 +128,15 @@ pub(crate) enum CliCommands { id: Uuid, }, + /// Delete a database: strip any checkpoints it pinned in parent databases, + /// then delete its own objects. Without --confirm, prints what it would + /// delete and does nothing. + DeleteDb { + /// Actually delete. Without it, this is a dry run. + #[arg(long)] + confirm: bool, + }, + /// List the current checkpoints of the db. ListCheckpoints { /// Optionally specify the name to filter the checkpoints. Note that name may not be unique diff --git a/slatedb-cli/src/main.rs b/slatedb-cli/src/main.rs index 08d5fd44ef..fd63d3cd9b 100644 --- a/slatedb-cli/src/main.rs +++ b/slatedb-cli/src/main.rs @@ -66,6 +66,7 @@ async fn main() -> Result<(), Box> { exec_refresh_checkpoint(&admin, id, lifetime).await?; } CliCommands::DeleteCheckpoint { id } => exec_delete_checkpoint(&admin, id).await?, + CliCommands::DeleteDb { confirm } => exec_delete_db(&admin, confirm).await?, CliCommands::ListCheckpoints { name } => exec_list_checkpoints(&admin, name).await?, CliCommands::RunGarbageCollection { resource, @@ -280,6 +281,20 @@ async fn exec_delete_checkpoint(admin: &Admin, id: Uuid) -> Result<(), Box Result<(), Box> { + let paths = admin.delete_db(confirm).await?; + if confirm { + println!("deleted {} objects", paths.len()); + } else { + println!("would delete {} objects:", paths.len()); + for path in &paths { + println!(" {path}"); + } + println!("rerun with --confirm to delete"); + } + Ok(()) +} + async fn exec_list_checkpoints( admin: &Admin, name_filter: Option, diff --git a/slatedb-common/Cargo.toml b/slatedb-common/Cargo.toml index 12d176e15b..0e5de9735c 100644 --- a/slatedb-common/Cargo.toml +++ b/slatedb-common/Cargo.toml @@ -23,3 +23,6 @@ test-util = ["tokio/test-util"] [dev-dependencies] tokio = { workspace = true, features = ["macros", "rt", "time", "test-util"] } + +[lints] +workspace = true diff --git a/slatedb-common/src/clock.rs b/slatedb-common/src/clock.rs index 2c38a6fe55..a5ae8babac 100644 --- a/slatedb-common/src/clock.rs +++ b/slatedb-common/src/clock.rs @@ -265,7 +265,7 @@ mod tests { #[tokio::test] #[cfg(feature = "test-util")] async fn test_mock_system_clock_set_now() { - let clock = std::sync::Arc::new(MockSystemClock::new()); + let clock = Arc::new(MockSystemClock::new()); // Test positive timestamp let positive_ts = 1625097600000i64; // 2021-07-01T00:00:00Z in milliseconds @@ -289,7 +289,7 @@ mod tests { #[tokio::test] #[cfg(feature = "test-util")] async fn test_mock_system_clock_advance() { - let clock = std::sync::Arc::new(MockSystemClock::new()); + let clock = Arc::new(MockSystemClock::new()); let initial_ts = 1000; // Set initial time @@ -310,7 +310,7 @@ mod tests { #[tokio::test] #[cfg(feature = "test-util")] async fn test_mock_system_clock_sleep() { - let clock = std::sync::Arc::new(MockSystemClock::new()); + let clock = Arc::new(MockSystemClock::new()); let initial_ts = 2000; // Set initial time @@ -352,7 +352,7 @@ mod tests { #[tokio::test] #[cfg(feature = "test-util")] async fn test_mock_system_clock_ticker() { - let clock = std::sync::Arc::new(MockSystemClock::new()); + let clock = Arc::new(MockSystemClock::new()); let tick_duration = Duration::from_millis(100); // Create a ticker @@ -388,7 +388,7 @@ mod tests { #[tokio::test(start_paused = true)] async fn test_default_system_clock_now() { - let clock = std::sync::Arc::new(DefaultSystemClock::new()); + let clock = Arc::new(DefaultSystemClock::new()); // Record initial time let initial_now = clock.now(); @@ -409,7 +409,7 @@ mod tests { #[tokio::test(start_paused = true)] #[cfg(feature = "test-util")] async fn test_default_system_clock_advance() { - let clock = std::sync::Arc::new(DefaultSystemClock::new()); + let clock = Arc::new(DefaultSystemClock::new()); let start = clock.now(); let duration = Duration::from_millis(500); clock.clone().advance(duration).await; @@ -424,7 +424,7 @@ mod tests { #[tokio::test(start_paused = true)] async fn test_default_system_clock_ticker() { - let clock = std::sync::Arc::new(DefaultSystemClock::new()); + let clock = Arc::new(DefaultSystemClock::new()); let tick_duration = Duration::from_millis(10); // Create a ticker diff --git a/slatedb-dst/src/actors/bank/mod.rs b/slatedb-dst/src/actors/bank/mod.rs index 5e76650089..6b6b76154f 100644 --- a/slatedb-dst/src/actors/bank/mod.rs +++ b/slatedb-dst/src/actors/bank/mod.rs @@ -1,6 +1,8 @@ mod auditor; mod transfer; +use std::mem::size_of; + use bytes::Bytes; use slatedb::config::{PutOptions, WriteOptions}; use slatedb::{Db, DbTransaction, Error, MergeOperator, MergeOperatorError}; @@ -8,7 +10,7 @@ use slatedb::{Db, DbTransaction, Error, MergeOperator, MergeOperatorError}; pub use self::auditor::{AuditorActor, BankAuditView}; pub use self::transfer::{TransferActor, TransferMode}; -const ACCUMULATOR_BYTES: usize = std::mem::size_of::(); +const ACCUMULATOR_BYTES: usize = size_of::(); /// Configuration for the deterministic bank workload. #[derive(Clone, Debug)] diff --git a/slatedb-dst/src/actors/fencer.rs b/slatedb-dst/src/actors/fencer.rs index 5b3be8b596..c7aeaa3837 100644 --- a/slatedb-dst/src/actors/fencer.rs +++ b/slatedb-dst/src/actors/fencer.rs @@ -80,7 +80,11 @@ impl Actor for DbFencerActor { let old_db = ctx.swap_db(next_db); // Verify the old DB is fenced. - match old_db.put(b"foo", b"bar").await { + let result = match old_db.put(b"foo", b"bar").await { + Ok(handle) => handle.await_durable().await, + Err(error) => Err(error), + }; + match result { Err(err) if matches!(err.kind(), ErrorKind::Closed(CloseReason::Fenced)) => {} result => panic!("old db was not fenced as expected [result={result:?}]"), } diff --git a/slatedb-dst/src/actors/workload.rs b/slatedb-dst/src/actors/workload.rs index a4221d59d7..cb78dda322 100644 --- a/slatedb-dst/src/actors/workload.rs +++ b/slatedb-dst/src/actors/workload.rs @@ -1,21 +1,20 @@ use std::collections::{BTreeMap, BTreeSet}; +use std::mem::size_of; use std::sync::atomic::{AtomicU64, Ordering}; use async_trait::async_trait; use bytes::Bytes; use log::info; use rand::RngCore; -use slatedb::config::{ - DurabilityLevel, MergeOptions, PutOptions, ReadOptions, ScanOptions, WriteOptions, -}; -use slatedb::{Error, MergeOperator, MergeOperatorError, WriteBatch}; +use slatedb::config::{DurabilityLevel, MergeOptions, PutOptions, ReadOptions, WriteOptions}; +use slatedb::{Error, IterationOrder, MergeOperator, MergeOperatorError, WriteBatch}; use tracing::instrument; -use crate::{Actor, ActorCtx}; +use crate::{utils::build_scan_options, Actor, ActorCtx}; use super::PROGRESS_LOG_INTERVAL; -const WORKLOAD_VALUE_VERSION_SIZE: usize = std::mem::size_of::(); +const WORKLOAD_VALUE_VERSION_SIZE: usize = size_of::(); /// Configuration for the mixed DST workload actor. #[derive(Clone, Debug)] @@ -351,12 +350,13 @@ async fn verify_scan( read_durability: DurabilityLevel, observed: &mut BTreeMap, ) -> Result<(), Error> { - let scan_options = ScanOptions::new().with_durability_filter(read_durability); + let scan_options = build_scan_options(ctx.rand(), read_durability); let mut iter = ctx .db() .scan_prefix_with_options(key_prefix.as_bytes(), .., &scan_options) .await?; let mut seen = BTreeSet::new(); + let mut previous_key: Option = None; while let Some(key_value) = iter.next().await? { assert!( @@ -375,6 +375,23 @@ async fn verify_scan( key_prefix, String::from_utf8_lossy(key_value.key.as_ref()).into_owned(), ); + if let Some(previous_key) = previous_key.as_ref() { + let is_ordered = match scan_options.order { + IterationOrder::Ascending => previous_key < &key_value.key, + IterationOrder::Descending => previous_key > &key_value.key, + }; + assert!( + is_ordered, + "workload scan returned keys out of order [name={}, step={}, key_prefix={}, order={:?}, previous_key={}, key={}]", + ctx.name(), + step, + key_prefix, + scan_options.order, + String::from_utf8_lossy(previous_key.as_ref()), + String::from_utf8_lossy(key_value.key.as_ref()), + ); + } + previous_key = Some(key_value.key.clone()); observe_present(ctx, step, &key_value.key, &key_value.value, observed); } diff --git a/slatedb-dst/src/deterministic_local_filesystem.rs b/slatedb-dst/src/deterministic_local_filesystem.rs index 13ffc7fb85..4b53e9fe06 100644 --- a/slatedb-dst/src/deterministic_local_filesystem.rs +++ b/slatedb-dst/src/deterministic_local_filesystem.rs @@ -1030,7 +1030,7 @@ async fn read_range_with_yields_from_file( Ok(buffer.into()) } fn convert_walkdir_result( - result: std::result::Result, + result: Result, ) -> object_store::Result> { match result { Ok(entry) => match symlink_metadata(entry.path()) { diff --git a/slatedb-dst/src/utils.rs b/slatedb-dst/src/utils.rs index 69c36de833..0bbd67b43a 100644 --- a/slatedb-dst/src/utils.rs +++ b/slatedb-dst/src/utils.rs @@ -4,11 +4,11 @@ use std::time::Duration; use rand::Rng; use slatedb::config::{ - CompactionWorkerOptions, CompactorOptions, CompressionCodec, DbReaderOptions, + CompactionWorkerOptions, CompactorOptions, CompressionCodec, DbReaderOptions, DurabilityLevel, GarbageCollectorDirectoryOptions, GarbageCollectorOptions, GarbageCollectorScheduleOptions, - SizeTieredCompactionSchedulerOptions, + ScanOptions, SizeTieredCompactionSchedulerOptions, }; -use slatedb::{DbRand, Settings}; +use slatedb::{DbRand, IterationOrder, Settings}; use tracing_subscriber::fmt::format::FmtSpan; use tracing_subscriber::EnvFilter; @@ -89,6 +89,22 @@ pub fn build_reader_options(rand: &DbRand) -> DbReaderOptions { } } +/// Builds randomized deterministic scan options for DST scenarios. +pub fn build_scan_options(rand: &DbRand, read_durability: DurabilityLevel) -> ScanOptions { + let mut rng = rand.rng(); + let read_ahead_options = [1, 4 * 1024, 64 * 1024, MIB_1]; + // Descending sorted-run iteration is currently broken. + // See https://github.com/slatedb/slatedb/pull/1995. + let order = IterationOrder::Ascending; + + ScanOptions::new() + .with_durability_filter(read_durability) + .with_read_ahead_bytes(read_ahead_options[rng.random_range(0..read_ahead_options.len())]) + .with_cache_blocks(rng.random_bool(0.5)) + .with_max_fetch_tasks(rng.random_range(1..=4)) + .with_order(order) +} + /// Builds randomized deterministic compactor options for DST scenarios. pub fn build_settings_compactor(rng: &mut impl Rng) -> CompactorOptions { let min_compaction_sources = rng.random_range(2..=4); diff --git a/slatedb-txn-obj/Cargo.toml b/slatedb-txn-obj/Cargo.toml index f15ec8b5b0..6675075d3e 100644 --- a/slatedb-txn-obj/Cargo.toml +++ b/slatedb-txn-obj/Cargo.toml @@ -24,3 +24,6 @@ test-util = [] [dev-dependencies] tempfile = { workspace = true } tokio = { workspace = true, features = ["macros", "rt", "time"] } + +[lints] +workspace = true diff --git a/slatedb-txn-obj/src/object_store.rs b/slatedb-txn-obj/src/object_store.rs index caf667621c..de0e7703ad 100644 --- a/slatedb-txn-obj/src/object_store.rs +++ b/slatedb-txn-obj/src/object_store.rs @@ -7,7 +7,7 @@ use bytes::Bytes; use futures::StreamExt; use log::{debug, error, warn}; use object_store::path::Path; -use object_store::Error::AlreadyExists; +use object_store::Error::{AlreadyExists, Precondition}; use object_store::{ Error, GetOptions, ObjectStore, ObjectStoreExt, PutMode, PutOptions, PutPayload, UpdateVersion, }; @@ -337,7 +337,7 @@ impl BoundaryObject for ObjectStoreBoundaryObject { return Ok(()); } // Try again if the boundary was concurrently updated by another process. - Err(Error::AlreadyExists { .. } | Error::Precondition { .. }) => { + Err(AlreadyExists { .. } | Precondition { .. }) => { // Refresh the cache so re-attempts always use the fresh boundary. self.read_boundary().await?; } diff --git a/slatedb/Cargo.toml b/slatedb/Cargo.toml index 56cbea36e1..4a0bcdd427 100644 --- a/slatedb/Cargo.toml +++ b/slatedb/Cargo.toml @@ -147,6 +147,10 @@ harness = false name = "scan_prefix_bench" harness = false +[[bench]] +name = "scan_prefix_large_sorted_run_bench" +harness = false + [[bench]] name = "block_iterator_v2" harness = false diff --git a/slatedb/benches/db_transaction.rs b/slatedb/benches/db_transaction.rs index 8954209b3b..5d26400f13 100644 --- a/slatedb/benches/db_transaction.rs +++ b/slatedb/benches/db_transaction.rs @@ -53,13 +53,13 @@ fn concat_merge_operator() -> Arc { fn put_options() -> PutOptions { PutOptions { - ttl: Ttl::ExpireAfter(3_600), + ttl: Ttl::ExpireAfterMillis(3_600_000), } } fn merge_options() -> MergeOptions { MergeOptions { - ttl: Ttl::ExpireAfter(3_600), + ttl: Ttl::ExpireAfterMillis(3_600_000), } } @@ -76,7 +76,7 @@ fn key(index: usize) -> Bytes { fn value(index: usize) -> Bytes { let mut value = vec![0; VALUE_SIZE]; - value[..std::mem::size_of::()].copy_from_slice(&index.to_le_bytes()); + value[..size_of::()].copy_from_slice(&index.to_le_bytes()); Bytes::from(value) } diff --git a/slatedb/benches/scan_prefix_bench.rs b/slatedb/benches/scan_prefix_bench.rs index f5b391a4da..93ca65241f 100644 --- a/slatedb/benches/scan_prefix_bench.rs +++ b/slatedb/benches/scan_prefix_bench.rs @@ -138,7 +138,7 @@ fn recency_scan_options() -> ScanOptions { async fn populate(db: &Db) { let write_opts = WriteOptions { await_durable: false, - seqnum: 0, + ..WriteOptions::default() }; let put_opts = PutOptions::default(); let mut next_version = [0u64; NUM_PREFIXES]; diff --git a/slatedb/benches/scan_prefix_large_sorted_run_bench.rs b/slatedb/benches/scan_prefix_large_sorted_run_bench.rs new file mode 100644 index 0000000000..f330ef9745 --- /dev/null +++ b/slatedb/benches/scan_prefix_large_sorted_run_bench.rs @@ -0,0 +1,242 @@ +// our microbenchmarks use pprof, but it doesn't work on windows +#![cfg(not(windows))] + +//! Measures how much of a prefix scan's setup cost comes from handing a +//! `SortedRun` to a scan iterator when the run holds far more SSTs than the +//! query range covers. +//! +//! Fixture: one sorted run of 1000 SSTs, one key per SST, disjoint and ordered +//! key ranges, with no memtable or L0 data left behind. Keys are `sr/000` +//! through `sr/999`, so each shorter prefix selects ten times as many SSTs. + +use std::collections::HashMap; +use std::sync::Arc; +use std::time::Duration; + +use bytes::Bytes; +use chrono::TimeDelta; +use criterion::{criterion_group, criterion_main, BenchmarkId, Criterion}; +use object_store::memory::InMemory; +use pprof::criterion::{Output, PProfProfiler}; +use slatedb::config::{ + CompactorOptions, DurabilityLevel, FlushOptions, FlushType, ScanOptions, Settings, +}; +use slatedb::Db; +use slatedb_common::clock::{DefaultSystemClock, SystemClock}; +use tokio::runtime::Runtime; + +/// Keys are `sr/000` through `sr/999`, one per SST. +const NUM_SSTS: usize = 1000; + +struct Case { + prefix: &'static str, + /// Keys under the prefix, which is also the number of rows a full prefix + /// scan returns. + keys: usize, + /// SST views `tables_covering_range` returns for the prefix range. + covering: usize, +} + +/// Each prefix drops one digit, so it selects ten times as many keys as the one +/// below it. +const CASES: [Case; 3] = [ + Case { + prefix: "sr/222", + keys: 1, + covering: 1, // the prefix is a single key, so it covers one SST view. + }, + Case { + prefix: "sr/22", + keys: 10, + covering: 11, // the prefix covers 11 SST views. + }, + Case { + prefix: "sr/2", + keys: 100, + covering: 101, // the prefix covers 101 SST views. + }, +]; + +fn make_key(idx: usize) -> Bytes { + Bytes::from(format!("sr/{idx:03}")) +} + +fn prefix_start(prefix: &'static str) -> Bytes { + Bytes::from_static(prefix.as_bytes()) +} + +/// Exclusive upper bound of the prefix range. Every prefix here ends in `2`, so +/// incrementing the last byte is enough. +fn prefix_end(prefix: &str) -> Bytes { + let mut end = prefix.as_bytes().to_vec(); + *end.last_mut().expect("prefix is non-empty") += 1; + Bytes::from(end) +} + +fn settings() -> Settings { + let mut scheduler_options = HashMap::new(); + // Fire exactly one compaction, and only once every L0 SST exists. + scheduler_options.insert("min_compaction_sources".to_string(), NUM_SSTS.to_string()); + scheduler_options.insert("max_compaction_sources".to_string(), NUM_SSTS.to_string()); + + Settings { + // Writers stall at these limits and the fixture parks NUM_SSTS in L0. + l0_max_ssts: NUM_SSTS + 16, + l0_max_ssts_per_key: NUM_SSTS + 16, + // One key per SST stays under the default min_filter_keys, so no SST + // carries a filter and every case is decided by key range overlap. + compactor_options: Some(CompactorOptions { + poll_interval: Duration::from_millis(50), + max_concurrent_compactions: 1, + // A trivial move preserves the one flush per SST topology. A + // rewriting compaction would choose its own output boundaries. + enable_trivial_move: true, + // No worker, so the coordinator either completes the compaction as + // a trivial move or the fixture assertion fails. + worker: None, + scheduler_options, + ..CompactorOptions::default() + }), + ..Settings::default() + } +} + +fn scan_options() -> ScanOptions { + ScanOptions { + cache_blocks: true, + durability_filter: DurabilityLevel::Remote, + ..ScanOptions::default() + } +} + +async fn populate(db: &Db) { + for idx in 0..NUM_SSTS { + let key = make_key(idx); + db.put(&key, &key).await.expect("put failed"); + db.flush_with_options(FlushOptions { + flush_type: FlushType::MemTable, + }) + .await + .expect("flush failed"); + } +} + +async fn await_single_sorted_run(db: &Db) { + let clock = DefaultSystemClock::new(); + let deadline = clock.now() + TimeDelta::seconds(300); + loop { + db.refresh_manifest() + .await + .expect("refresh_manifest failed"); + let manifest = db.manifest(); + let l0 = manifest.l0().len(); + let runs = manifest.compacted(); + let ssts = runs + .first() + .map_or(0, |run| run.tables_covering_range(..).len()); + if l0 == 0 && runs.len() == 1 && ssts == NUM_SSTS { + return; + } + assert!( + clock.now() < deadline, + "compaction never produced one sorted run of {NUM_SSTS} ssts (l0={l0}, runs={}, ssts={ssts})", + runs.len() + ); + clock.sleep(Duration::from_millis(50)).await; + } +} + +async fn count_prefix_rows(db: &Db, prefix: &'static str) -> usize { + let mut iter = db + .scan_prefix_with_options(prefix_start(prefix), .., &scan_options()) + .await + .expect("scan_prefix failed"); + let mut count = 0usize; + while iter.next().await.expect("iterator next failed").is_some() { + count += 1; + } + count +} + +async fn validate_fixture(db: &Db) { + let manifest = db.manifest(); + let run = manifest + .compacted() + .first() + .expect("fixture has one sorted run"); + for case in &CASES { + let covering = run + .tables_covering_range(prefix_start(case.prefix)..prefix_end(case.prefix)) + .len(); + assert_eq!( + covering, case.covering, + "prefix {} covers {covering} sst views", + case.prefix + ); + let rows = count_prefix_rows(db, case.prefix).await; + assert_eq!( + rows, case.keys, + "prefix {} scanned {rows} rows", + case.prefix + ); + } +} + +async fn build_fixture() -> Db { + let store = Arc::new(InMemory::new()); + let db = Db::builder("/bench/sorted_run", store) + .with_settings(settings()) + .build() + .await + .expect("failed to build db"); + populate(&db).await; + await_single_sorted_run(&db).await; + validate_fixture(&db).await; + db +} + +fn bench_scan_prefix_large_sorted_run(c: &mut Criterion) { + let runtime = Runtime::new().expect("failed to create runtime"); + let db = runtime.block_on(build_fixture()); + + let scan_opts = scan_options(); + let mut group = c.benchmark_group("scan_prefix_large_sorted_run"); + + for case in &CASES { + let label = format!("K={}", case.keys); + + group.bench_function(BenchmarkId::new("first_entry", &label), |b| { + b.to_async(&runtime).iter(|| async { + let mut iter = db + .scan_prefix_with_options(prefix_start(case.prefix), .., &scan_opts) + .await + .expect("scan_prefix failed"); + let entry = iter.next().await.expect("iterator next failed"); + assert!(entry.is_some()); + }); + }); + + group.bench_function(BenchmarkId::new("first_entry_by_recency", &label), |b| { + b.to_async(&runtime).iter(|| async { + let mut iter = db + .scan_prefix_by_recency_with_options(prefix_start(case.prefix), &scan_opts) + .await + .expect("scan_prefix_by_recency failed"); + let entry = iter.next_entry().await.expect("iterator next_entry failed"); + assert!(entry.is_some()); + }); + }); + } + + group.finish(); + runtime.block_on(async { db.close().await.expect("close failed") }); +} + +criterion_group! { + name = benches; + config = Criterion::default() + .with_profiler(PProfProfiler::new(100, Output::Protobuf)); + targets = bench_scan_prefix_large_sorted_run +} + +criterion_main!(benches); diff --git a/slatedb/benches/write_batch.rs b/slatedb/benches/write_batch.rs index 2bb2c11ea6..b9bc7c85de 100644 --- a/slatedb/benches/write_batch.rs +++ b/slatedb/benches/write_batch.rs @@ -56,13 +56,13 @@ fn size_sum_merge_operator() -> Arc { fn put_options() -> PutOptions { PutOptions { - ttl: Ttl::ExpireAfter(3_600), + ttl: Ttl::ExpireAfterMillis(3_600_000), } } fn merge_options() -> MergeOptions { MergeOptions { - ttl: Ttl::ExpireAfter(3_600), + ttl: Ttl::ExpireAfterMillis(3_600_000), } } @@ -72,7 +72,7 @@ fn key(index: usize) -> Bytes { fn value(index: usize) -> Bytes { let mut value = vec![0; VALUE_SIZE]; - value[..std::mem::size_of::()].copy_from_slice(&index.to_le_bytes()); + value[..size_of::()].copy_from_slice(&index.to_le_bytes()); Bytes::from(value) } diff --git a/slatedb/src/admin.rs b/slatedb/src/admin.rs index 7555ba7895..aef082f0bf 100644 --- a/slatedb/src/admin.rs +++ b/slatedb/src/admin.rs @@ -1,5 +1,6 @@ pub use crate::db::builder::CloneBuilder; pub use crate::db::builder::CloneSourceSpec; +use std::collections::BTreeSet; use crate::checkpoint::{Checkpoint, CheckpointCreateResult}; use crate::compactions_store::CompactionsStore; @@ -19,8 +20,9 @@ use crate::seq_tracker::FindOption; use crate::utils::IdGenerator; use bytes::Bytes; use chrono::{DateTime, Utc}; +use futures::StreamExt; use object_store::path::Path; -use object_store::ObjectStore; +use object_store::{ObjectStore, ObjectStoreExt}; use rand::RngCore; use slatedb_common::DbRand; use std::env; @@ -34,6 +36,7 @@ use uuid::Uuid; pub use crate::db::builder::AdminBuilder; use crate::merge_operator::MergeOperatorType; +use crate::wal::WalAdmin; use slatedb_txn_obj::TransactionalObject; /// An Admin struct for SlateDB administration operations. @@ -56,6 +59,7 @@ pub struct Admin { pub(crate) compaction_filter_supplier: Option>, pub(crate) merge_operator: Option, + pub(crate) wal_admin: Arc, } impl Admin { @@ -103,11 +107,21 @@ impl Admin { .map_err(crate::Error::from)?; let mut manifests = Vec::with_capacity(manifest_metadata.len()); for metadata in manifest_metadata { - let manifest = manifest_store - .read_manifest(metadata.id) + match manifest_store + .try_read_manifest(metadata.id) .await - .map_err(crate::Error::from)?; - manifests.push(VersionedManifest::from_manifest(metadata.id, manifest)); + .map_err(crate::Error::from)? + { + Some(manifest) => { + manifests.push(VersionedManifest::from_manifest(metadata.id, manifest)) + } + // Deleted after LIST by a concurrent GC + // See https://github.com/slatedb/slatedb/issues/1215 for more details. + None => log::warn!( + "listed manifest missing on read, skipping [id={}]", + metadata.id + ), + } } Ok(manifests) } @@ -283,6 +297,7 @@ impl Admin { self.object_stores.store_of(ObjectStoreType::Main).clone(), ) .with_system_clock(self.system_clock.clone()) + .with_wal_gc(self.wal_admin.garbage_collector(&self.path)) .with_wal_object_store(self.object_stores.store_of(ObjectStoreType::Wal).clone()) .with_options(gc_opts) .with_seed(self.rand.rng().next_u64()) @@ -316,6 +331,7 @@ impl Admin { self.object_stores.store_of(ObjectStoreType::Main).clone(), ) .with_system_clock(self.system_clock.clone()) + .with_wal_gc(self.wal_admin.garbage_collector(&self.path)) .with_wal_object_store(self.object_stores.store_of(ObjectStoreType::Wal).clone()) .with_options(gc_opts) .with_seed(self.rand.rng().next_u64()) @@ -559,6 +575,122 @@ impl Admin { .map_err(Into::into) } + /// Deletes a database, stripping any checkpoints it pinned in parent + /// databases (a clone) before removing its own objects. Works for plain and + /// cloned dbs alike: a plain db just has no parent checkpoints to strip. + /// + /// Without `confirm` this is a dry run: it returns every object it *would* + /// delete and touches nothing. Pass `confirm` to actually delete. + /// + /// The delete writes a `.deleting` marker while the manifest still proves + /// this is a slatedb dir, then removes everything else, then the marker. If a + /// prior run crashed mid-delete, the leftover marker lets a rerun finish the + /// job. A `confirm` delete of a dir with neither a manifest nor a marker is + /// refused, so a fat-fingered `--path` can't wipe an unrelated directory. + /// Idempotent. + pub async fn delete_db(&self, confirm: bool) -> Result, crate::Error> { + let main = self.retrying_store(ObjectStoreType::Main); + + if !confirm { + return self.list_prefix(&main).await; + } + + let marker = self.path.clone().join(".deleting"); + let marker_exists = main.get(&marker).await.map(|_| true).or_else(|e| match e { + object_store::Error::NotFound { .. } => Ok(false), + other => Err(SlateDBError::from(other)), + })?; + + let manifest = self.manifest_store().try_read_latest_manifest().await?; + + // No manifest and no marker means we never proved this is a slatedb dir. + // If there are objects under the prefix, it may be a fat-fingered path we + // must not wipe, so refuse. If empty, there's nothing to delete anyway + // (also the already-deleted no-op), so fall through to a clean return. + if manifest.is_none() + && !marker_exists + && !collect_prefix(&main, &self.path).await?.is_empty() + { + return Err(SlateDBError::InvalidDBState.into()); + } + + // Strip the checkpoints this db pinned in each parent. Needs the + // manifest, which a resumed (marker-only) run may no longer have. + if let Some(manifest) = manifest.as_ref() { + for external_db in manifest.external_dbs() { + let Some(final_checkpoint_id) = external_db.final_checkpoint_id else { + continue; + }; + let parent_store = Arc::new(ManifestStore::new( + &Path::from(external_db.path.as_str()), + self.retrying_store(ObjectStoreType::Main), + )); + let mut parent = + match StoredManifest::load(parent_store, self.system_clock.clone()).await { + Ok(parent) => parent, + // parent already deleted: no checkpoint left to strip, skip it + Err(SlateDBError::LatestTransactionalObjectVersionMissing) => continue, + Err(e) => return Err(e.into()), + }; + parent.delete_checkpoint(final_checkpoint_id).await?; + } + } + + // Commit the intent to delete while the manifest still proves this dir. + if !marker_exists { + main.put(&marker, Bytes::new().into()) + .await + .map_err(SlateDBError::from)?; + } + + // Delete everything but the marker, then the marker last, so a crash in + // between leaves the marker to prove a rerun should finish. + let mut deleted = self.delete_prefix(&main, Some(&marker)).await?; + deleted.extend( + self.wal_admin + .delete_wal(&self.path, false) + .await + .map_err(SlateDBError::from)?, + ); + main.delete(&marker).await.map_err(SlateDBError::from)?; + deleted.push(marker.to_string()); + Ok(deleted) + } + + /// Lists every object under this db's path prefix across the main and WAL stores. + async fn list_prefix(&self, main: &Arc) -> Result, crate::Error> { + let paths = collect_prefix(main, &self.path).await?; + // track the dry run paths in a set since the WAL and db may overlap + let mut paths = paths.iter().map(Path::to_string).collect::>(); + let wal_paths = self + .wal_admin + .delete_wal(&self.path, true) + .await + .map_err(SlateDBError::from)?; + for wp in wal_paths { + paths.insert(wp); + } + Ok(paths.into_iter().collect()) + } + + /// Deletes every object under this db's path prefix in the given store, + /// skipping `keep` if set. Returns the deleted paths. + async fn delete_prefix( + &self, + store: &Arc, + keep: Option<&Path>, + ) -> Result, crate::Error> { + let mut deleted = Vec::new(); + for path in collect_prefix(store, &self.path).await? { + if Some(&path) == keep { + continue; + } + store.delete(&path).await.map_err(SlateDBError::from)?; + deleted.push(path); + } + Ok(deleted.into_iter().map(|p| p.to_string()).collect()) + } + /// Returns the timestamp or sequence from the latest manifest's sequence tracker. /// When `round_up` is true, uses the next higher value; otherwise the previous one. pub async fn get_timestamp_for_sequence( @@ -639,6 +771,8 @@ impl Admin { /// [`CloneBuilder::with_source`]: each source must carry its own per-source range so that /// [`crate::manifest::Manifest::cloned_from_union`] sees non-overlapping effective ranges. /// + /// Segmented sources only have to be non-overlapping within each segment that they share. + /// /// # Examples /// /// ``` @@ -670,7 +804,7 @@ impl Admin { source, self.retrying_store(ObjectStoreType::Main), ) - .with_wal_object_store(self.retrying_store(ObjectStoreType::Wal)) + .with_wal_admin(self.wal_admin.clone()) } /// Creates a new builder for an admin client at the given path. @@ -815,6 +949,24 @@ pub fn load_gcp() -> Result, crate::Error> { })?) as Arc) } +/// Collects every object path under `prefix` in the given store. +async fn collect_prefix( + store: &Arc, + prefix: &Path, +) -> Result, crate::Error> { + let mut listing = store.list(Some(prefix)); + let mut paths = Vec::new(); + while let Some(meta) = listing + .next() + .await + .transpose() + .map_err(SlateDBError::from)? + { + paths.push(meta.location); + } + Ok(paths) +} + #[cfg(test)] mod tests { use crate::admin::{load_object_store_from_env, AdminBuilder}; @@ -1389,6 +1541,291 @@ mod tests { assert!(manifest.is_ok(), "cloned manifest should exist"); } + #[tokio::test] + async fn test_delete_db_removes_checkpoint_from_parent() { + use crate::admin::CloneSourceSpec; + use crate::config::CheckpointOptions; + use crate::manifest::store::{ManifestStore, StoredManifest}; + use crate::Db; + + let object_store: Arc = Arc::new(InMemory::new()); + let system_clock = Arc::new(DefaultSystemClock::new()); + let parent_path = Path::from("/tmp/test_cleanup_parent"); + let clone_path = Path::from("/tmp/test_cleanup_clone"); + + let parent_db = Db::open(parent_path.clone(), object_store.clone()) + .await + .unwrap(); + parent_db.close().await.unwrap(); + + // An unrelated checkpoint in the parent that cleanup must not touch. + let parent_admin = AdminBuilder::new(parent_path.clone(), object_store.clone()).build(); + let unrelated = parent_admin + .create_detached_checkpoint(&CheckpointOptions::default()) + .await + .unwrap() + .id; + + let clone_admin = AdminBuilder::new(clone_path.clone(), object_store.clone()).build(); + clone_admin + .create_clone_builder_from_source(CloneSourceSpec::new(parent_path.clone())) + .build() + .await + .expect("clone should succeed"); + + // The checkpoint the clone pinned in the parent. + let clone_ms = Arc::new(ManifestStore::new(&clone_path, object_store.clone())); + let clone_stored = StoredManifest::load(clone_ms, system_clock.clone()) + .await + .unwrap(); + let pinned = clone_stored.manifest().external_dbs[0] + .final_checkpoint_id + .expect("clone pins a final_checkpoint_id in the parent"); + + let read_parent_checkpoints = || { + let object_store = object_store.clone(); + let system_clock = system_clock.clone(); + let parent_path = parent_path.clone(); + async move { + let ms = Arc::new(ManifestStore::new(&parent_path, object_store)); + let stored = StoredManifest::load(ms, system_clock).await.unwrap(); + stored + .manifest() + .core + .checkpoints + .iter() + .map(|c| c.id) + .collect::>() + } + }; + + let before = read_parent_checkpoints().await; + assert!( + before.contains(&pinned), + "parent should have pinned checkpoint before cleanup" + ); + assert!( + before.contains(&unrelated), + "parent should have unrelated checkpoint" + ); + + clone_admin + .delete_db(true) + .await + .expect("delete should succeed"); + + let after = read_parent_checkpoints().await; + assert!( + !after.contains(&pinned), + "pinned checkpoint should be gone from parent" + ); + assert!( + after.contains(&unrelated), + "unrelated checkpoint should remain" + ); + } + + #[tokio::test] + async fn test_delete_db_deletes_clone_after_parent_already_gone() { + use crate::admin::CloneSourceSpec; + use crate::Db; + use futures::StreamExt; + use object_store::ObjectStoreExt; + + let object_store: Arc = Arc::new(InMemory::new()); + let parent_path = Path::from("/tmp/test_delete_orphan_parent"); + let clone_path = Path::from("/tmp/test_delete_orphan_clone"); + + Db::open(parent_path.clone(), object_store.clone()) + .await + .unwrap() + .close() + .await + .unwrap(); + + let clone_admin = AdminBuilder::new(clone_path.clone(), object_store.clone()).build(); + clone_admin + .create_clone_builder_from_source(CloneSourceSpec::new(parent_path.clone())) + .build() + .await + .expect("clone should succeed"); + + // Wipe the parent out from under the clone. The clone still names it in + // external_dbs with a pinned checkpoint, but the parent manifest is gone. + let mut parent_listing = object_store.list(Some(&parent_path)); + while let Some(meta) = parent_listing.next().await { + object_store.delete(&meta.unwrap().location).await.unwrap(); + } + + // A missing parent means there is no checkpoint left to strip, so the + // clone must still be deletable rather than wedged on a Data error. + clone_admin + .delete_db(true) + .await + .expect("clone should delete even with parent already gone"); + assert_eq!( + object_store.list(Some(&clone_path)).count().await, + 0, + "clone objects should be gone" + ); + } + + #[tokio::test] + async fn test_delete_db_deletes_own_objects_only_with_confirm() { + use crate::admin::CloneSourceSpec; + use crate::Db; + use futures::StreamExt; + + let object_store: Arc = Arc::new(InMemory::new()); + let parent_path = Path::from("/tmp/test_delete_confirm_parent"); + let clone_path = Path::from("/tmp/test_delete_confirm_clone"); + + Db::open(parent_path.clone(), object_store.clone()) + .await + .unwrap() + .close() + .await + .unwrap(); + + let clone_admin = AdminBuilder::new(clone_path.clone(), object_store.clone()).build(); + clone_admin + .create_clone_builder_from_source(CloneSourceSpec::new(parent_path.clone())) + .build() + .await + .expect("clone should succeed"); + + let count_under = |prefix: Path| { + let object_store = object_store.clone(); + async move { object_store.list(Some(&prefix)).count().await } + }; + + let initial = count_under(clone_path.clone()).await; + assert!(initial > 0, "clone should have objects"); + + // confirm = false is a dry run: it reports what it would delete and + // touches nothing. + let would_delete = clone_admin + .delete_db(false) + .await + .expect("dry run should succeed"); + assert_eq!( + would_delete.len(), + initial, + "dry run should report every object under the prefix" + ); + assert_eq!( + count_under(clone_path.clone()).await, + initial, + "dry run must not delete anything" + ); + + // confirm = true deletes the clone's own objects. + clone_admin + .delete_db(true) + .await + .expect("delete should succeed"); + assert_eq!( + count_under(clone_path.clone()).await, + 0, + "clone objects should be gone after confirm" + ); + + // Idempotent: a second run over an already-deleted db is a clean no-op. + clone_admin + .delete_db(true) + .await + .expect("second delete should be a no-op"); + } + + #[tokio::test] + async fn test_delete_db_finishes_partial_deletion_via_marker() { + use crate::admin::CloneSourceSpec; + use crate::Db; + use futures::StreamExt; + use object_store::ObjectStoreExt; + + let object_store: Arc = Arc::new(InMemory::new()); + let parent_path = Path::from("/tmp/test_delete_partial_parent"); + let clone_path = Path::from("/tmp/test_delete_partial_clone"); + + Db::open(parent_path.clone(), object_store.clone()) + .await + .unwrap() + .close() + .await + .unwrap(); + + let clone_admin = AdminBuilder::new(clone_path.clone(), object_store.clone()).build(); + clone_admin + .create_clone_builder_from_source(CloneSourceSpec::new(parent_path.clone())) + .build() + .await + .expect("clone should succeed"); + + // Simulate a crash mid-delete: the marker was written, then the manifests + // (and some objects) were removed, but the marker and other objects remain. + object_store + .put( + &clone_path.clone().join(".deleting"), + bytes::Bytes::new().into(), + ) + .await + .unwrap(); + let manifest_prefix = Path::from("/tmp/test_delete_partial_clone/manifest"); + let mut listing = object_store.list(Some(&manifest_prefix)); + while let Some(meta) = listing.next().await { + object_store.delete(&meta.unwrap().location).await.unwrap(); + } + let count_under = |prefix: Path| { + let object_store = object_store.clone(); + async move { object_store.list(Some(&prefix)).count().await } + }; + assert!( + count_under(clone_path.clone()).await > 0, + "leftover clone objects should remain after partial deletion" + ); + + // The marker proves this is a real slatedb dir, so delete resumes and + // finishes the job even though the manifest is already gone. + clone_admin + .delete_db(true) + .await + .expect("delete should finish partial deletion"); + assert_eq!( + count_under(clone_path.clone()).await, + 0, + "leftover objects, including the marker, should be gone" + ); + } + + #[tokio::test] + async fn test_delete_db_refuses_without_manifest_or_marker() { + use futures::StreamExt; + use object_store::ObjectStoreExt; + + let object_store: Arc = Arc::new(InMemory::new()); + let dir = Path::from("/tmp/test_delete_fat_finger"); + + // A directory that is NOT a slatedb dir: no manifest, no .deleting marker. + // This stands in for a fat-fingered --path. + object_store + .put( + &dir.clone().join("important.txt"), + bytes::Bytes::from_static(b"keepme").into(), + ) + .await + .unwrap(); + + let admin = AdminBuilder::new(dir.clone(), object_store.clone()).build(); + admin + .delete_db(true) + .await + .expect_err("delete must refuse a dir with no manifest and no marker"); + + let count = object_store.list(Some(&dir)).count().await; + assert_eq!(count, 1, "the unrelated object must be left untouched"); + } + #[cfg(feature = "wal_disable")] #[tokio::test] async fn test_create_clone_with_multiple_sources() { @@ -1472,4 +1909,144 @@ mod tests { "clone should have an external database for each parent" ); } + + #[cfg(feature = "wal_disable")] + #[tokio::test] + async fn test_delete_db_removes_checkpoints_from_all_parents() { + use crate::config::{PutOptions, Settings, WriteOptions}; + use crate::manifest::store::{ManifestStore, StoredManifest}; + use crate::{admin::CloneSourceSpec, Db}; + use uuid::Uuid; + + let object_store: Arc = Arc::new(InMemory::new()); + let system_clock = Arc::new(DefaultSystemClock::new()); + let parent_path1 = Path::from("/tmp/test_cleanup_multi_parent1"); + let parent_path2 = Path::from("/tmp/test_cleanup_multi_parent2"); + let clone_path = Path::from("/tmp/test_cleanup_multi_clone"); + + let settings = Settings { + wal_enabled: false, + ..Settings::default() + }; + let write_opts = WriteOptions { + await_durable: false, + ..Default::default() + }; + + // Two parents with disjoint single-key SSTs (the union path rejects overlaps). + for (path, key) in [(&parent_path1, b"a"), (&parent_path2, b"z")] { + let db = Db::builder(path.clone(), object_store.clone()) + .with_settings(settings.clone()) + .build() + .await + .unwrap(); + db.put_with_options(key, b"1", &PutOptions::default(), &write_opts) + .await + .unwrap(); + db.close().await.unwrap(); + } + + let clone_admin = AdminBuilder::new(clone_path.clone(), object_store.clone()).build(); + clone_admin + .create_clone_builder_from_source(CloneSourceSpec::new(parent_path1.clone())) + .with_source(CloneSourceSpec::new(parent_path2.clone())) + .build() + .await + .expect("clone with multiple sources should succeed"); + + // Collect the checkpoint each parent got pinned with. + let clone_ms = Arc::new(ManifestStore::new(&clone_path, object_store.clone())); + let clone_stored = StoredManifest::load(clone_ms, system_clock.clone()) + .await + .unwrap(); + let pinned: Vec<(String, Uuid)> = clone_stored + .manifest() + .external_dbs + .iter() + .map(|e| (e.path.clone(), e.final_checkpoint_id.unwrap())) + .collect(); + assert_eq!(pinned.len(), 2); + + clone_admin + .delete_db(true) + .await + .expect("delete should succeed"); + + for (parent_path, checkpoint_id) in pinned { + let ms = Arc::new(ManifestStore::new( + &parent_path.into(), + object_store.clone(), + )); + let stored = StoredManifest::load(ms, system_clock.clone()) + .await + .unwrap(); + assert!( + !stored + .manifest() + .core + .checkpoints + .iter() + .any(|c| c.id == checkpoint_id), + "pinned checkpoint should be removed from every parent" + ); + } + } +} + +#[cfg(test)] +mod gc_tolerant_list_manifests_tests { + use crate::admin::AdminBuilder; + use crate::manifest::store::{ManifestStore, StoredManifest}; + use crate::manifest::ManifestCore; + use crate::test_utils::FlakyObjectStore; + use object_store::memory::InMemory; + use object_store::path::Path; + use object_store::ObjectStore; + use slatedb_common::clock::DefaultSystemClock; + use std::sync::Arc; + + #[tokio::test] + async fn test_list_manifests_skips_manifest_gced_between_list_and_read() { + let inner: Arc = Arc::new(InMemory::new()); + let flaky = Arc::new(FlakyObjectStore::new(inner, 0)); + let store: Arc = flaky.clone(); + let path = Path::from("/tmp/test_gc_tolerant_list_manifests"); + + let manifest_store = Arc::new(ManifestStore::new(&path, store.clone())); + let mut sm = StoredManifest::create_new_db( + manifest_store, + ManifestCore::new(), + Arc::new(DefaultSystemClock::new()), + ) + .await + .unwrap(); + sm.update(sm.prepare_dirty().unwrap()).await.unwrap(); + sm.update(sm.prepare_dirty().unwrap()).await.unwrap(); + + let admin = AdminBuilder::new(path.clone(), store.clone()).build(); + let ids: Vec = admin + .list_manifests(..) + .await + .unwrap() + .iter() + .map(|vm| vm.id()) + .collect(); + assert_eq!(ids, vec![1, 2, 3]); + + // manifest 1 is listed, but missing on read + flaky.with_get_not_found_failures(1); + + let ids: Vec = admin + .list_manifests(..) + .await + .unwrap() + .iter() + .map(|vm| vm.id()) + .collect(); + assert_eq!( + ids, + vec![2, 3], + "a manifest GC'd mid-listing should be skipped, not fail the operation" + ); + } } diff --git a/slatedb/src/batch.rs b/slatedb/src/batch.rs index f29c9c712b..4ec07cbc53 100644 --- a/slatedb/src/batch.rs +++ b/slatedb/src/batch.rs @@ -16,6 +16,7 @@ use bytes::Bytes; use smallvec::{smallvec, SmallVec}; use std::collections::{BTreeMap, BTreeSet, HashMap, HashSet}; use std::iter::Peekable; +use std::mem::size_of; use std::ops::RangeBounds; use std::sync::Arc; @@ -146,7 +147,7 @@ impl DisjointMergeDiscriminators { .iter() .fold(0, |bytes, value| bytes.saturating_add(value.0.len())), DisjointMergeDiscriminatorSet::Tokens(tokens) => { - tokens.len().saturating_mul(core::mem::size_of::()) + tokens.len().saturating_mul(size_of::()) } } } @@ -672,7 +673,7 @@ impl WriteBatch { &self, seq: u64, now: i64, - default_ttl: Option, + default_ttl_millis: Option, merger: Option, extractor: Option<&dyn PrefixExtractor>, ) -> Result<(Vec, BTreeSet, u64), SlateDBError> { @@ -683,7 +684,7 @@ impl WriteBatch { IterationOrder::Ascending, seq, Some(now), - default_ttl, + default_ttl_millis, )); if self.has_merge_ops() { if let Some(ref merge_operator) = merger { @@ -742,12 +743,12 @@ pub mod benches { batch: &WriteBatch, seq: u64, now: i64, - default_ttl: Option, + default_ttl_millis: Option, merger: Option>, extractor: Option<&dyn PrefixExtractor>, ) -> Result<(Vec, BTreeSet, u64), Error> { batch - .extract_entries(seq, now, default_ttl, merger, extractor) + .extract_entries(seq, now, default_ttl_millis, merger, extractor) .await .map_err(Into::into) } @@ -773,11 +774,11 @@ impl WriteBatchIterator { op: &WriteOp, seq: u64, now: Option, - default_ttl: Option, + default_ttl_millis: Option, ) -> RowEntry { let expire_ts = match (op, now) { - (WriteOp::Put(_, opts), Some(now)) => opts.expire_ts_from(default_ttl, now), - (WriteOp::Merge(_, opts), Some(now)) => opts.expire_ts_from(default_ttl, now), + (WriteOp::Put(_, opts), Some(now)) => opts.expire_ts_from(default_ttl_millis, now), + (WriteOp::Merge(_, opts), Some(now)) => opts.expire_ts_from(default_ttl_millis, now), _ => None, }; op.to_row_entry(key, seq, now, expire_ts) @@ -789,21 +790,21 @@ impl WriteBatchIterator { ordering: IterationOrder, seq: u64, now: Option, - default_ttl: Option, + default_ttl_millis: Option, ) -> Self { let entries: Vec = match ordering { IterationOrder::Ascending => batch .ops .range(range) .flat_map(|(key, ops)| ops.iter().rev().map(move |op| (key, op))) - .map(|(key, op)| Self::write_op_to_row_entry(key, op, seq, now, default_ttl)) + .map(|(key, op)| Self::write_op_to_row_entry(key, op, seq, now, default_ttl_millis)) .collect(), IterationOrder::Descending => batch .ops .range(range) .rev() .flat_map(|(key, ops)| ops.iter().rev().map(move |op| (key, op))) - .map(|(key, op)| Self::write_op_to_row_entry(key, op, seq, now, default_ttl)) + .map(|(key, op)| Self::write_op_to_row_entry(key, op, seq, now, default_ttl_millis)) .collect(), }; @@ -818,21 +819,23 @@ impl WriteBatchIterator { #[async_trait] impl RowEntryIterator for WriteBatchIterator { - async fn init(&mut self) -> Result<(), crate::error::SlateDBError> { + async fn init(&mut self) -> Result<(), SlateDBError> { Ok(()) } - async fn next(&mut self) -> Result, crate::error::SlateDBError> { + async fn next(&mut self) -> Result, SlateDBError> { Ok(self.iter.next()) } - async fn seek(&mut self, next_key: &[u8]) -> Result<(), crate::error::SlateDBError> { + async fn seek(&mut self, next_key: &[u8]) -> Result<(), SlateDBError> { while let Some(entry) = self.iter.peek() { if match self.ordering { IterationOrder::Ascending => entry.key.as_ref() < next_key, IterationOrder::Descending => entry.key.as_ref() > next_key, } { self.iter.next(); + // Keep in-memory seeking cooperative. + tokio::task::coop::consume_budget().await; } else { break; } @@ -1403,7 +1406,7 @@ mod tests { // Given: an empty WriteBatch and custom merge options let mut batch = WriteBatch::new(); let merge_options = MergeOptions { - ttl: Ttl::ExpireAfter(3600), // 1 hour + ttl: Ttl::ExpireAfterMillis(3_600_000), // 1 hour }; // When: adding a merge operation with custom options @@ -1415,7 +1418,7 @@ mod tests { match op { WriteOp::Merge(value, options) => { assert_eq!(value.as_ref(), b"value1"); - assert_eq!(options.ttl, Ttl::ExpireAfter(3600)); + assert_eq!(options.ttl, Ttl::ExpireAfterMillis(3_600_000)); } _ => panic!("Expected Merge operation"), } @@ -1874,8 +1877,7 @@ mod tests { ]; assert_iterator(&mut iter, expected).await; - let merge_operator = Some(std::sync::Arc::new(StringConcatMergeOperator) - as crate::merge_operator::MergeOperatorType); + let merge_operator = Some(Arc::new(StringConcatMergeOperator) as MergeOperatorType); let (result, _, _) = batch .extract_entries(100, 1000, None, merge_operator, None) .await @@ -1949,8 +1951,7 @@ mod tests { batch.merge(b"key1", b"merge3"); // When: extracting entries with a merge operator - let merge_operator = Some(std::sync::Arc::new(StringConcatMergeOperator) - as crate::merge_operator::MergeOperatorType); + let merge_operator = Some(Arc::new(StringConcatMergeOperator) as MergeOperatorType); let (result, _, _) = batch .extract_entries(100, 1000, None, merge_operator, None) .await @@ -1972,19 +1973,18 @@ mod tests { b"key1", b"a", &MergeOptions { - ttl: Ttl::ExpireAfter(3600), + ttl: Ttl::ExpireAfterMillis(3600), }, ); batch.merge_with_options( b"key1", b"b", &MergeOptions { - ttl: Ttl::ExpireAt(4600), + ttl: Ttl::ExpireAtMillis(4600), }, ); - let merge_operator = Some(std::sync::Arc::new(StringConcatMergeOperator) - as crate::merge_operator::MergeOperatorType); + let merge_operator = Some(Arc::new(StringConcatMergeOperator) as MergeOperatorType); let (result, _, _) = batch .extract_entries(100, 1000, None, merge_operator, None) .await @@ -2006,19 +2006,18 @@ mod tests { b"key1", b"a", &MergeOptions { - ttl: Ttl::ExpireAfter(3600), + ttl: Ttl::ExpireAfterMillis(3600), }, ); batch.merge_with_options( b"key1", b"b", &MergeOptions { - ttl: Ttl::ExpireAfter(7200), + ttl: Ttl::ExpireAfterMillis(7200), }, ); - let merge_operator = Some(std::sync::Arc::new(StringConcatMergeOperator) - as crate::merge_operator::MergeOperatorType); + let merge_operator = Some(Arc::new(StringConcatMergeOperator) as MergeOperatorType); let err = batch .extract_entries(100, 1000, None, merge_operator, None) .await @@ -2045,19 +2044,18 @@ mod tests { b"key1", b"a", &MergeOptions { - ttl: Ttl::ExpireAfter(3600), + ttl: Ttl::ExpireAfterMillis(3600), }, ); batch.merge_with_options( b"key2", b"b", &MergeOptions { - ttl: Ttl::ExpireAfter(7200), + ttl: Ttl::ExpireAfterMillis(7200), }, ); - let merge_operator = Some(std::sync::Arc::new(StringConcatMergeOperator) - as crate::merge_operator::MergeOperatorType); + let merge_operator = Some(Arc::new(StringConcatMergeOperator) as MergeOperatorType); let (result, _, _) = batch .extract_entries(100, 1000, None, merge_operator, None) .await @@ -2078,12 +2076,11 @@ mod tests { b"key1", b"a", &MergeOptions { - ttl: Ttl::ExpireAfter(7200), + ttl: Ttl::ExpireAfterMillis(7200), }, ); - let merge_operator = Some(std::sync::Arc::new(StringConcatMergeOperator) - as crate::merge_operator::MergeOperatorType); + let merge_operator = Some(Arc::new(StringConcatMergeOperator) as MergeOperatorType); let err = batch .extract_entries(100, 1000, None, merge_operator, None) .await @@ -2100,12 +2097,11 @@ mod tests { b"key1", b"a", &MergeOptions { - ttl: Ttl::ExpireAfter(7200), + ttl: Ttl::ExpireAfterMillis(7200), }, ); - let merge_operator = Some(std::sync::Arc::new(StringConcatMergeOperator) - as crate::merge_operator::MergeOperatorType); + let merge_operator = Some(Arc::new(StringConcatMergeOperator) as MergeOperatorType); let err = batch .extract_entries(100, 1000, None, merge_operator, None) .await @@ -2123,8 +2119,7 @@ mod tests { batch.merge(b"key1", b"merge2"); // When: extracting entries with a merge operator - let merge_operator = Some(std::sync::Arc::new(StringConcatMergeOperator) - as crate::merge_operator::MergeOperatorType); + let merge_operator = Some(Arc::new(StringConcatMergeOperator) as MergeOperatorType); let (result, _, _) = batch .extract_entries(100, 1000, None, merge_operator, None) .await @@ -2149,8 +2144,7 @@ mod tests { batch.merge(b"key1", b"merge2"); // When: extracting entries with a merge operator - let merge_operator = Some(std::sync::Arc::new(StringConcatMergeOperator) - as crate::merge_operator::MergeOperatorType); + let merge_operator = Some(Arc::new(StringConcatMergeOperator) as MergeOperatorType); let (result, _, _) = batch .extract_entries(100, 1000, None, merge_operator, None) .await diff --git a/slatedb/src/batch_write.rs b/slatedb/src/batch_write.rs index 1749177d46..7a8afe9d43 100644 --- a/slatedb/src/batch_write.rs +++ b/slatedb/src/batch_write.rs @@ -29,13 +29,9 @@ use async_trait::async_trait; use fail_parallel::fail_point; use futures::stream::BoxStream; use futures::{FutureExt, StreamExt}; -use log::warn; use std::sync::Arc; -use std::time::Duration; use tracing::instrument; -use std::collections::BTreeSet; - use crate::config::WriteOptions; use crate::db_state::DbState; use crate::db_transaction::DbTransaction; @@ -47,7 +43,7 @@ use crate::wal::{FlushResultFuture, WalWriter}; use crate::{batch::WriteBatch, db::DbInner, db::WriteHandle, error::SlateDBError}; use bytes::Bytes; use parking_lot::RwLockWriteGuard; -use slatedb_common::clock::SystemClock; +use std::collections::BTreeSet; use tokio::sync::oneshot; pub(crate) const WRITE_BATCH_TASK_NAME: &str = "writer"; @@ -102,7 +98,6 @@ impl std::fmt::Debug for BatchWriterMessage { pub(crate) struct WriteBatchEventHandler { db_inner: Arc, - is_first_write: bool, wal_writer: Option>, } @@ -110,7 +105,6 @@ impl WriteBatchEventHandler { pub(crate) fn new(db_inner: Arc, wal_writer: Option>) -> Self { Self { db_inner, - is_first_write: true, wal_writer, } } @@ -126,17 +120,11 @@ impl MessageHandler for WriteBatchEventHandler { done, txn, }) => { + let wal_writer = self.wal_writer.as_deref_mut(); let result = self .db_inner - .write_batch( - batch, - &options, - txn.as_ref(), - self.wal_writer.as_mut(), - self.is_first_write, - ) + .write_batch(batch, &options, txn.as_ref(), wal_writer) .await; - self.is_first_write = false; match result { Ok(write_result) => { let _ = done.send(write_result); @@ -201,13 +189,12 @@ impl MessageHandler for WriteBatchEventHandler { impl DbInner { #[allow(clippy::panic)] #[instrument(level = "trace", skip_all, fields(batch_size = batch.op_count()))] - async fn write_batch( - &self, + async fn write_batch<'a>( + &'a self, batch: WriteBatch, options: &WriteOptions, txn: Option<&DbTransaction>, - wal_writer: Option<&mut Box>, - is_first_write: bool, + mut wal_writer: Option<&mut (dyn WalWriter + 'static)>, ) -> Result { let _options = options; #[cfg(not(dst))] @@ -246,7 +233,7 @@ impl DbInner { .extract_entries( commit_seq, now, - self.settings.default_ttl, + self.settings.default_ttl_millis, self.flush_merge_operator.clone(), self.segment_extractor.as_deref(), ) @@ -265,7 +252,7 @@ impl DbInner { return Ok(Err(error)); } - if let Some(wal_writer) = wal_writer { + if let Some(wal_writer) = wal_writer.as_mut() { assert!(self.wal_enabled); // WAL entries must be appended to the wal buffer atomically. Otherwise, // the WAL buffer might flush the entries in the middle of the batch, which @@ -278,17 +265,7 @@ impl DbInner { self.write_entries_to_memtable(entries, touched_segments); } else { assert!(!self.wal_enabled); - // if WAL is disabled, we just write the entries to memtable. - let watcher = self.write_entries_to_memtable(entries, touched_segments); - // if this is the first write and the WAL is disabled, make sure users are flushing - // their memtables in a timely manner. - if is_first_write && options.await_durable { - let this_watcher = watcher.clone(); - let this_clock = self.system_clock.clone(); - tokio::spawn(async move { - monitor_first_write(this_watcher, this_clock).await; - }); - } + self.write_entries_to_memtable(entries, touched_segments); }; // increment memtable_write_bytes by the size of the keys and values inserted into the memtable // after merge operators and overwrites are collapsed @@ -326,38 +303,47 @@ impl DbInner { self.record_memtable_sequence(commit_seq); // maybe freeze the memtable. - self.maybe_freeze_current_memtable()?; + self.maybe_freeze_current_memtable(wal_writer.as_deref())?; - let write_handle = WriteHandle::new(commit_seq, now); + let write_handle = + WriteHandle::new_with_waiter(commit_seq, now, self.status_manager.durability_waiter()); Ok(Ok(write_handle)) } - fn maybe_freeze_current_memtable(&self) -> Result<(), SlateDBError> { - let replay_after_wal_id = self.wal_observer.status()?.last_flushed_wal_id; - let mut guard = self.state.write(); - let meta = guard.memtable().metadata(); - - let last_freeze_wal_id = guard - .state() - .imm_memtable - .front() - .map(|imm| imm.recent_flushed_wal_id()) - .unwrap_or(guard.state().core().replay_after_wal_id); - - let l0_sst_size_est = self - .table_store - .estimate_encoded_size_compacted(meta.entry_num, meta.entries_size_in_bytes); + fn maybe_freeze_current_memtable( + &self, + wal_writer: Option<&dyn WalWriter>, + ) -> Result<(), SlateDBError> { + // extract the information required to make a freeze decision under a read-lock first + // so we don't call into `WalWriter` while holding a lock + let (l0_sst_size_est, last_freeze_wal_id) = { + let guard = self.state.read(); + let meta = guard.memtable().metadata(); + + let last_freeze_wal_id = guard + .state() + .imm_memtable + .front() + .map(|imm| imm.recent_flushed_wal_id()) + .unwrap_or(guard.state().core().replay_after_wal_id); + + let l0_sst_size_est = self + .table_store + .estimate_encoded_size_compacted(meta.entry_num, meta.entries_size_in_bytes); + + (l0_sst_size_est, last_freeze_wal_id) + }; - let wal_id_gap = replay_after_wal_id - .checked_sub(last_freeze_wal_id) - .ok_or_else(|| SlateDBError::InvalidDBState)?; + let wal_should_flush_memtable = wal_writer + .is_some_and(|wal_writer| wal_writer.should_flush_memtable(last_freeze_wal_id)); - if wal_id_gap < self.settings.max_wal_flushes_before_l0_flush - && l0_sst_size_est < self.settings.l0_sst_size_bytes - { + if !wal_should_flush_memtable && l0_sst_size_est < self.settings.l0_sst_size_bytes { return Ok(()); } + + let replay_after_wal_id = self.wal_observer.status()?.last_flushed_wal_id; + let mut guard = self.state.write(); self.freeze_current_memtable_with_state_guard(&mut guard, replay_after_wal_id); Ok(()) } @@ -405,7 +391,7 @@ impl DbInner { &self, freeze_memtable: bool, ) -> Result<(), SlateDBError> { - let (done, rx) = tokio::sync::oneshot::channel(); + let (done, rx) = oneshot::channel(); self.write_notifier .send(BatchWriterMessage::Flush(BatchWriterFlush { freeze_memtable, @@ -521,21 +507,6 @@ fn check_segment_prefix_antichain( Ok(()) } -async fn monitor_first_write( - mut watcher: WatchableOnceCellReader>, - system_clock: Arc, -) { - tokio::select! { - _ = watcher.await_value() => {} - _ = system_clock.sleep(Duration::from_secs(5)) => { - warn!("First write not durable after 5 seconds and WAL is disabled. \ - SlateDB does not automatically flush memtables until `l0_sst_size_bytes` \ - is reached. If writer is single threaded or has low throughput, the \ - applications must call `flush` to ensure durability in a timely manner."); - } - } -} - #[cfg(test)] mod tests { use super::*; @@ -597,11 +568,8 @@ mod tests { fn test_message( batch: WriteBatch, options: WriteOptions, - ) -> ( - BatchWriterMessage, - tokio::sync::oneshot::Receiver, - ) { - let (done, rx) = tokio::sync::oneshot::channel(); + ) -> (BatchWriterMessage, oneshot::Receiver) { + let (done, rx) = oneshot::channel(); ( BatchWriterMessage::WriteBatch(WriteBatchRequest { batch, @@ -613,31 +581,6 @@ mod tests { ) } - #[tokio::test] - async fn test_is_first_write_set_false_after_first_write() { - let object_store = Arc::new(InMemory::new()); - let db = Db::open( - "/tmp/test_is_first_write_set_false_after_first_write", - object_store, - ) - .await - .unwrap(); - - let wal_writer = Box::new(FakeWalWriter::new(0)); - let mut handler = WriteBatchEventHandler::new(db.inner.clone(), Some(wal_writer)); - assert!(handler.is_first_write); - - let mut batch = WriteBatch::new(); - batch.put(b"key", b"value"); - - let (msg, done_rx) = test_message(batch, WriteOptions::default()); - handler.handle(msg).await.unwrap(); - - let result = done_rx.await.unwrap(); - assert!(result.is_ok()); - assert!(!handler.is_first_write); - } - #[tokio::test] async fn test_append_error_notifies_caller_and_fails_handler() { let object_store = Arc::new(InMemory::new()); @@ -677,7 +620,7 @@ mod tests { .unwrap(); let wal_writer = Box::new(FailingWalWriter::new(FailingWalOperation::Flush)); let mut handler = WriteBatchEventHandler::new(db.inner.clone(), Some(wal_writer)); - let (done, done_rx) = tokio::sync::oneshot::channel(); + let (done, done_rx) = oneshot::channel(); let msg = BatchWriterMessage::Flush(BatchWriterFlush { freeze_memtable: false, done, @@ -709,6 +652,8 @@ mod tests { let (msg, done_rx) = test_message( batch, WriteOptions { + #[cfg(dst)] + now: 0, seqnum: 42, ..Default::default() }, @@ -753,6 +698,8 @@ mod tests { let (msg, done_rx) = test_message( batch, WriteOptions { + #[cfg(dst)] + now: 0, seqnum: 1, ..Default::default() }, diff --git a/slatedb/src/blob.rs b/slatedb/src/blob.rs index 4986e356c7..00d8dfb377 100644 --- a/slatedb/src/blob.rs +++ b/slatedb/src/blob.rs @@ -19,6 +19,7 @@ pub(crate) struct BytesBlob { } impl BytesBlob { + #[allow(dead_code)] pub(crate) fn new(bytes: Bytes) -> Self { Self { bytes } } diff --git a/slatedb/src/bytes_generator.rs b/slatedb/src/bytes_generator.rs index 07d9c85b9a..fb746f4c28 100644 --- a/slatedb/src/bytes_generator.rs +++ b/slatedb/src/bytes_generator.rs @@ -30,7 +30,7 @@ impl OrderedBytesGenerator { } pub(crate) fn next(&mut self) -> Bytes { - let mut result = BytesMut::with_capacity(self.bytes.len() + std::mem::size_of::()); + let mut result = BytesMut::with_capacity(self.bytes.len() + size_of::()); result.put_slice(self.bytes.as_slice()); result.put(self.suffix.as_ref()); self.increment(); diff --git a/slatedb/src/bytes_range.rs b/slatedb/src/bytes_range.rs index cbda899f42..3d3a318b7d 100644 --- a/slatedb/src/bytes_range.rs +++ b/slatedb/src/bytes_range.rs @@ -21,69 +21,69 @@ pub trait ByteRangeBounds { fn bound_as_bytes>(bound: Bound<&K>) -> Bound<&[u8]> { match bound { - Bound::Included(k) => Bound::Included(k.as_ref()), - Bound::Excluded(k) => Bound::Excluded(k.as_ref()), - Bound::Unbounded => Bound::Unbounded, + Included(k) => Included(k.as_ref()), + Excluded(k) => Excluded(k.as_ref()), + Unbounded => Unbounded, } } impl> ByteRangeBounds for Range { fn start_bound(&self) -> Bound<&[u8]> { - Bound::Included(self.start.as_ref()) + Included(self.start.as_ref()) } fn end_bound(&self) -> Bound<&[u8]> { - Bound::Excluded(self.end.as_ref()) + Excluded(self.end.as_ref()) } } impl> ByteRangeBounds for RangeInclusive { fn start_bound(&self) -> Bound<&[u8]> { - bound_as_bytes(std::ops::RangeBounds::start_bound(self)) + bound_as_bytes(RangeBounds::start_bound(self)) } fn end_bound(&self) -> Bound<&[u8]> { - bound_as_bytes(std::ops::RangeBounds::end_bound(self)) + bound_as_bytes(RangeBounds::end_bound(self)) } } impl> ByteRangeBounds for RangeFrom { fn start_bound(&self) -> Bound<&[u8]> { - Bound::Included(self.start.as_ref()) + Included(self.start.as_ref()) } fn end_bound(&self) -> Bound<&[u8]> { - Bound::Unbounded + Unbounded } } impl> ByteRangeBounds for RangeTo { fn start_bound(&self) -> Bound<&[u8]> { - Bound::Unbounded + Unbounded } fn end_bound(&self) -> Bound<&[u8]> { - Bound::Excluded(self.end.as_ref()) + Excluded(self.end.as_ref()) } } impl> ByteRangeBounds for RangeToInclusive { fn start_bound(&self) -> Bound<&[u8]> { - Bound::Unbounded + Unbounded } fn end_bound(&self) -> Bound<&[u8]> { - Bound::Included(self.end.as_ref()) + Included(self.end.as_ref()) } } impl ByteRangeBounds for RangeFull { fn start_bound(&self) -> Bound<&[u8]> { - Bound::Unbounded + Unbounded } fn end_bound(&self) -> Bound<&[u8]> { - Bound::Unbounded + Unbounded } } @@ -100,17 +100,17 @@ impl ByteRangeBounds for BytesRange { impl> ByteRangeBounds for (Bound, Bound) { fn start_bound(&self) -> Bound<&[u8]> { match &self.0 { - Bound::Included(v) => Bound::Included(v.as_ref()), - Bound::Excluded(v) => Bound::Excluded(v.as_ref()), - Bound::Unbounded => Bound::Unbounded, + Included(v) => Included(v.as_ref()), + Excluded(v) => Excluded(v.as_ref()), + Unbounded => Unbounded, } } fn end_bound(&self) -> Bound<&[u8]> { match &self.1 { - Bound::Included(v) => Bound::Included(v.as_ref()), - Bound::Excluded(v) => Bound::Excluded(v.as_ref()), - Bound::Unbounded => Bound::Unbounded, + Included(v) => Included(v.as_ref()), + Excluded(v) => Excluded(v.as_ref()), + Unbounded => Unbounded, } } } @@ -301,7 +301,7 @@ impl BytesRange { pub(crate) fn as_point(&self) -> Option<&Bytes> { match (RangeBounds::start_bound(self), RangeBounds::end_bound(self)) { - (Bound::Included(start), Bound::Included(end)) if start == end => Some(start), + (Included(start), Included(end)) if start == end => Some(start), _ => None, } } @@ -315,8 +315,7 @@ pub(crate) mod tests { use bytes::Bytes; use proptest::{prop_assert, prop_assert_eq, proptest}; - use std::ops::Bound; - use std::ops::Bound::{Included, Unbounded}; + use std::ops::Bound::{Excluded, Included, Unbounded}; use std::ops::RangeBounds; #[test] @@ -348,22 +347,17 @@ pub(crate) mod tests { fn test_byte_range_bounds_for_common_shapes() { let full = BytesRange::from_prefix_and_subrange(b"p", ..); assert_eq!(full.start_bound(), Included(&Bytes::from_static(b"p"))); - assert_eq!(full.end_bound(), Bound::Excluded(&Bytes::from_static(b"q"))); + assert_eq!(full.end_bound(), Excluded(&Bytes::from_static(b"q"))); let range = BytesRange::from_prefix_and_subrange(b"", b"a".to_vec()..=b"b".to_vec()); assert_eq!(range.start_bound(), Included(&Bytes::from_static(b"a"))); assert_eq!(range.end_bound(), Included(&Bytes::from_static(b"b"))); - let tuple = BytesRange::from_prefix_and_subrange( - b"ab", - (Bound::Excluded(&b"x"[..]), Bound::Included(&b"y"[..])), - ); + let tuple = + BytesRange::from_prefix_and_subrange(b"ab", (Excluded(&b"x"[..]), Included(&b"y"[..]))); assert_eq!( tuple, - BytesRange::from_prefix_and_subrange( - b"ab", - (Bound::Excluded(&b"x"[..]), Bound::Included(&b"y"[..])) - ) + BytesRange::from_prefix_and_subrange(b"ab", (Excluded(&b"x"[..]), Included(&b"y"[..]))) ); } @@ -372,8 +366,8 @@ pub(crate) mod tests { let range = BytesRange::from_prefix_and_subrange(b"user1:", &b"0005"[..]..&b"0042"[..]); let start = Bytes::from("user1:0005"); let end = Bytes::from("user1:0042"); - assert_eq!(range.start_bound(), Bound::Included(&start)); - assert_eq!(range.end_bound(), Bound::Excluded(&end)); + assert_eq!(range.start_bound(), Included(&start)); + assert_eq!(range.end_bound(), Excluded(&end)); } #[test] @@ -381,8 +375,8 @@ pub(crate) mod tests { let range = BytesRange::from_prefix_and_subrange(b"ab", ..=&b"x"[..]); let start = Bytes::from("ab"); let end = Bytes::from("abx"); - assert_eq!(range.start_bound(), Bound::Included(&start)); - assert_eq!(range.end_bound(), Bound::Included(&end)); + assert_eq!(range.start_bound(), Included(&start)); + assert_eq!(range.end_bound(), Included(&end)); } #[test] @@ -391,18 +385,15 @@ pub(crate) mod tests { let range = BytesRange::from_prefix_and_subrange(b"ab", &b"x"[..]..); let start = Bytes::from("abx"); let end = Bytes::from("ac"); - assert_eq!(range.start_bound(), Bound::Included(&start)); - assert_eq!(range.end_bound(), Bound::Excluded(&end)); + assert_eq!(range.start_bound(), Included(&start)); + assert_eq!(range.end_bound(), Excluded(&end)); } #[test] fn test_from_prefix_and_subrange_excluded_start() { - let range = BytesRange::from_prefix_and_subrange( - b"ab", - (Bound::Excluded(&b"x"[..]), Bound::Unbounded), - ); + let range = BytesRange::from_prefix_and_subrange(b"ab", (Excluded(&b"x"[..]), Unbounded)); let start = Bytes::from("abx"); - assert_eq!(range.start_bound(), Bound::Excluded(&start)); + assert_eq!(range.start_bound(), Excluded(&start)); } #[test] @@ -410,8 +401,8 @@ pub(crate) mod tests { let prefix = vec![0xff, 0xff]; let range = BytesRange::from_prefix_and_subrange(&prefix, &b"a"[..]..); let start = Bytes::from(vec![0xff, 0xff, b'a']); - assert_eq!(range.start_bound(), Bound::Included(&start)); - assert_eq!(range.end_bound(), Bound::Unbounded); + assert_eq!(range.start_bound(), Included(&start)); + assert_eq!(range.end_bound(), Unbounded); } #[test] @@ -453,8 +444,8 @@ pub(crate) mod tests { let range = BytesRange::from_prefix(b"ab"); let start = Bytes::from("ab"); let end = Bytes::from("ac"); - assert_eq!(range.start_bound(), Bound::Included(&start)); - assert_eq!(range.end_bound(), Bound::Excluded(&end)); + assert_eq!(range.start_bound(), Included(&start)); + assert_eq!(range.end_bound(), Excluded(&end)); } #[test] @@ -462,15 +453,15 @@ pub(crate) mod tests { let prefix = vec![0xff, 0xff]; let range = BytesRange::from_prefix(&prefix); let start = Bytes::from(prefix); - assert_eq!(range.start_bound(), Bound::Included(&start)); - assert_eq!(range.end_bound(), Bound::Unbounded); + assert_eq!(range.start_bound(), Included(&start)); + assert_eq!(range.end_bound(), Unbounded); } #[test] fn test_from_prefix_allows_empty_prefix() { let range = BytesRange::from_prefix(b""); - assert_eq!(range.start_bound(), Bound::Unbounded); - assert_eq!(range.end_bound(), Bound::Unbounded); + assert_eq!(range.start_bound(), Unbounded); + assert_eq!(range.end_bound(), Unbounded); } #[test] @@ -567,14 +558,14 @@ pub(crate) mod tests { #[test] #[should_panic(expected = "Range must be non-empty")] fn test_new_with_start_larger_than_end_panics() { - let start = Bound::Included(Bytes::from("z")); - let end = Bound::Included(Bytes::from("a")); + let start = Included(Bytes::from("z")); + let end = Included(Bytes::from("a")); BytesRange::new(start, end); } #[test] fn test_empty_included_start_bound_is_valid_and_contains_all_keys() { - let range = BytesRange::new(Bound::Included(Bytes::new()), Bound::Unbounded); + let range = BytesRange::new(Included(Bytes::new()), Unbounded); assert!(range.contains(&Bytes::new())); // b"" <= b"" holds for Included assert!(range.contains(&Bytes::from("a"))); assert!(range.contains(&Bytes::from("z"))); @@ -582,7 +573,7 @@ pub(crate) mod tests { #[test] fn test_empty_excluded_start_bound_is_valid_and_contains_all_keys() { - let range = BytesRange::new(Bound::Excluded(Bytes::new()), Bound::Unbounded); + let range = BytesRange::new(Excluded(Bytes::new()), Unbounded); assert!(!range.contains(&Bytes::new())); // b"" < b"" is false for Excluded assert!(range.contains(&Bytes::from("a"))); assert!(range.contains(&Bytes::from("z"))); diff --git a/slatedb/src/cached_object_store/object_store.rs b/slatedb/src/cached_object_store/object_store.rs index 79763d4a41..1e39852055 100644 --- a/slatedb/src/cached_object_store/object_store.rs +++ b/slatedb/src/cached_object_store/object_store.rs @@ -249,7 +249,7 @@ impl CachedObjectStore { // Second pass: load the selected files in bounded parallelism and cache them. let degree_of_parallelism = 32; - let _result = build_concurrent(files_to_load.into_iter(), degree_of_parallelism, |path| { + let _result = build_concurrent(files_to_load, degree_of_parallelism, |path| { let this = self.clone(); async move { match this @@ -367,8 +367,8 @@ impl CachedObjectStore { async fn cached_put_opts( &self, location: &Path, - payload: object_store::PutPayload, - opts: object_store::PutOptions, + payload: PutPayload, + opts: PutOptions, ) -> object_store::Result { // The per-call tag decides whether this write is cached. let tag = ObjectStoreCallTag::from_extensions(&opts.extensions); @@ -390,7 +390,7 @@ impl CachedObjectStore { // Convert PutPayload to stream and save parts to cache. let entry = self.cache_storage.entry(location, self.part_size_bytes); - let stream = stream::iter(payload.into_iter()).map(Ok::); + let stream = stream::iter(payload).map(Ok::); // Save parts, ignoring errors (cache failures must not fail the PUT). self.save_parts_stream(entry.as_ref(), stream, 0).await.ok(); @@ -1440,7 +1440,7 @@ mod tests { inner .put( &location, - PutPayload::from_bytes(bytes::Bytes::from_static(b"hello world")), + PutPayload::from_bytes(Bytes::from_static(b"hello world")), ) .await .unwrap(); @@ -1460,10 +1460,7 @@ mod tests { .expect("cache miss should fetch from inner store"); assert!(result.extensions.get::().is_some()); - assert_eq!( - result.bytes().await.unwrap(), - bytes::Bytes::from_static(b"hello") - ); + assert_eq!(result.bytes().await.unwrap(), Bytes::from_static(b"hello")); } #[tokio::test] @@ -1473,7 +1470,7 @@ mod tests { inner .put( &location, - PutPayload::from_bytes(bytes::Bytes::from_static(b"hello")), + PutPayload::from_bytes(Bytes::from_static(b"hello")), ) .await .unwrap(); diff --git a/slatedb/src/cached_object_store/storage_fs.rs b/slatedb/src/cached_object_store/storage_fs.rs index bbac906116..6ebf5d9974 100644 --- a/slatedb/src/cached_object_store/storage_fs.rs +++ b/slatedb/src/cached_object_store/storage_fs.rs @@ -193,11 +193,7 @@ impl FsCacheStorage { #[async_trait::async_trait] impl LocalCacheStorage for FsCacheStorage { - fn entry( - &self, - location: &object_store::path::Path, - part_size: usize, - ) -> Box { + fn entry(&self, location: &Path, part_size: usize) -> Box { Box::new(FsCacheEntry { root_folder: self.root_folder.clone(), location: location.clone(), @@ -1227,7 +1223,7 @@ impl FsCacheEvictorInner { // a specific index. Returns None if no available index exists. fn pick_random_available_index( &self, - rng: &mut impl rand::Rng, + rng: &mut impl Rng, keys: &[std::path::PathBuf], picked: &HashSet, exclude_idx: Option, @@ -1351,7 +1347,7 @@ mod tests { .prefix("objstore_cache_test_evictor_") .tempdir() .unwrap(); - let recorder = slatedb_common::metrics::MetricsRecorderHelper::noop(); + let recorder = MetricsRecorderHelper::noop(); let evictor = FsCacheEvictorInner::new( temp_dir.path().to_path_buf(), @@ -1379,7 +1375,7 @@ mod tests { .await; assert_eq!(evicted, 2048); - let file_paths = walkdir::WalkDir::new(temp_dir.path()) + let file_paths = WalkDir::new(temp_dir.path()) .into_iter() .map(|entry| entry.unwrap().file_name().to_string_lossy().to_string()) .collect::>(); @@ -1392,7 +1388,7 @@ mod tests { .prefix("objstore_cache_test_evictor_backpressure_") .tempdir() .unwrap(); - let recorder = slatedb_common::metrics::MetricsRecorderHelper::noop(); + let recorder = MetricsRecorderHelper::noop(); let evictor = FsCacheEvictor::new( temp_dir.path().to_path_buf(), @@ -1438,7 +1434,7 @@ mod tests { .prefix("objstore_cache_test_evictor_") .tempdir() .unwrap(); - let recorder = slatedb_common::metrics::MetricsRecorderHelper::noop(); + let recorder = MetricsRecorderHelper::noop(); let evictor = Arc::new(FsCacheEvictorInner::new( temp_dir.path().to_path_buf(), 1024 * 2, @@ -1469,7 +1465,7 @@ mod tests { .prefix("objstore_cache_test_evictor_") .tempdir() .unwrap(); - let recorder = slatedb_common::metrics::MetricsRecorderHelper::noop(); + let recorder = MetricsRecorderHelper::noop(); let evictor = Arc::new(FsCacheEvictorInner::new( temp_dir.path().to_path_buf(), @@ -1497,7 +1493,7 @@ mod tests { .prefix("objstore_cache_usage_snapshot_") .tempdir() .unwrap(); - let recorder = slatedb_common::metrics::MetricsRecorderHelper::noop(); + let recorder = MetricsRecorderHelper::noop(); let usage = Arc::new(FsCacheUsage { initialized: AtomicBool::new(false), used_bytes: AtomicU64::new(0), @@ -1571,7 +1567,7 @@ mod tests { .prefix("objstore_cache_usage_updates_") .tempdir() .unwrap(); - let recorder = slatedb_common::metrics::MetricsRecorderHelper::noop(); + let recorder = MetricsRecorderHelper::noop(); let storage = FsCacheStorage::new( temp_dir.path().to_path_buf(), None, @@ -1658,7 +1654,7 @@ mod tests { .prefix("objstore_cache_test_pick_") .tempdir() .unwrap(); - let recorder = slatedb_common::metrics::MetricsRecorderHelper::noop(); + let recorder = MetricsRecorderHelper::noop(); let evictor = FsCacheEvictorInner::new( temp_dir.path().to_path_buf(), 1024, diff --git a/slatedb/src/checkpoint.rs b/slatedb/src/checkpoint.rs index 21a6dee911..9451675402 100644 --- a/slatedb/src/checkpoint.rs +++ b/slatedb/src/checkpoint.rs @@ -54,7 +54,7 @@ mod tests { use crate::block_cache_policy::BlockCachePolicy; use crate::checkpoint::Checkpoint; use crate::checkpoint::CheckpointCreateResult; - use crate::config::{CheckpointOptions, CheckpointScope, Settings}; + use crate::config::{CheckpointOptions, CheckpointScope, PutOptions, Settings, WriteOptions}; use crate::db::Db; use crate::db_state::{SsTableId, SsTableView}; use crate::format::sst::SsTableFormat; @@ -343,8 +343,16 @@ mod tests { .await .unwrap(); - db.put(b"k1", b"v1").await.unwrap(); - db.put(b"k2", b"v2").await.unwrap(); + let write_options = WriteOptions { + await_durable: false, + ..Default::default() + }; + db.put_with_options(b"k1", b"v1", &PutOptions::default(), &write_options) + .await + .unwrap(); + db.put_with_options(b"k2", b"v2", &PutOptions::default(), &write_options) + .await + .unwrap(); let checkpoint = db .create_checkpoint(CheckpointScope::All, &CheckpointOptions::default()) diff --git a/slatedb/src/clone.rs b/slatedb/src/clone.rs index 7f8a03644e..4e74af30b6 100644 --- a/slatedb/src/clone.rs +++ b/slatedb/src/clone.rs @@ -3,19 +3,16 @@ use crate::checkpoint::Checkpoint; use crate::config::CheckpointOptions; use crate::db::builder::CloneSourceSpec; -use crate::db_state::SsTableId; use crate::error::SlateDBError; use crate::error::SlateDBError::CheckpointMissing; use crate::manifest::store::{ManifestStore, StoredManifest}; -use crate::manifest::{Manifest, ManifestCore, ProjectionConfig}; -use crate::object_stores::ObjectStoreType::{Main, Wal}; -use crate::object_stores::ObjectStores; -use crate::paths::PathResolver; +use crate::manifest::{Manifest, ProjectionConfig, VersionedManifest}; use crate::utils::IdGenerator; +use crate::wal::WalAdmin; use bytes::Bytes; use fail_parallel::{fail_point, FailPointRegistry}; use object_store::path::Path; -use object_store::{ObjectStore, ObjectStoreExt}; +use object_store::ObjectStore; use slatedb_common::clock::SystemClock; use slatedb_common::DbRand; use std::ops::RangeBounds; @@ -35,10 +32,22 @@ pub(crate) type SegmentFilterFn = Arc bool + Send + Sync>; pub(crate) type SegmentProjectionFn = Arc Result + Send + Sync>; +struct CopyWalParams { + from_path: Path, + from_manifest: VersionedManifest, + to_path: Path, +} + +struct CreateCloneManifestResult { + clone_manifest: StoredManifest, + copy_wal_params: Option, +} + pub(crate) async fn create_clone, R: RangeBounds + Clone>( clone_sources: Vec>, clone_path: P, - object_stores: ObjectStores, + object_store: Arc, + wal_admin: Arc, fp_registry: Arc, system_clock: Arc, rand: Arc, @@ -48,40 +57,37 @@ pub(crate) async fn create_clone, R: RangeBounds + Clone>( ) -> Result<(), SlateDBError> { let clone_path = clone_path.into(); - validate_clone_source_specs(clone_sources.clone(), clone_path.clone())?; + validate_clone_source_specs(&clone_sources, &clone_path)?; - let mut clone_manifest = create_clone_manifest( + let CreateCloneManifestResult { + mut clone_manifest, + copy_wal_params, + } = create_clone_manifest( clone_path.clone(), - clone_sources.clone(), - object_stores.store_of(Main).clone(), - object_stores.store_of(Wal).clone(), + clone_sources, + object_store, system_clock.clone(), rand, fp_registry.clone(), projection_range, segment_filter, segment_projection, + wal_admin.as_ref(), ) .await?; if !clone_manifest.db_state().initialized { - // Copy WAL SSTs from all sources - WAL is only supported for single source - // this invariant is enforced in create_clone_manifest() - if clone_sources.len() == 1 { - for source in &clone_sources { - let parent_path = source.path.clone(); - copy_wal_ssts( - object_stores.store_of(Wal).clone(), - clone_manifest.db_state(), - &parent_path, - &clone_path, - fp_registry.clone(), - ) - .await?; - } - } + let (replay_after_wal_id, wal_id_last_seen) = match copy_wal_params { + Some(params) => copy_wal(wal_admin.as_ref(), params).await?, + None => (0, 0), + }; + let next_wal_sst_id = wal_id_last_seen + .checked_add(1) + .ok_or(SlateDBError::InvalidDBState)?; let mut dirty = clone_manifest.prepare_dirty()?; + dirty.value.core.replay_after_wal_id = replay_after_wal_id; + dirty.value.core.next_wal_sst_id = next_wal_sst_id; dirty.value.core.initialized = true; clone_manifest.update(dirty).await?; } @@ -93,17 +99,17 @@ async fn create_clone_manifest + Clone>( clone_path: Path, source_specs: Vec>, object_store: Arc, - wal_object_store: Arc, system_clock: Arc, rand: Arc, #[allow(unused)] fp_registry: Arc, projection_range: Option, segment_filter: Option, segment_projection: Option, -) -> Result { + wal_admin: &dyn WalAdmin, +) -> Result { let clone_manifest_store = Arc::new(ManifestStore::new(&clone_path, object_store.clone())); - let clone_manifest = + let (clone_manifest, copy_wal_params) = match StoredManifest::try_load(clone_manifest_store.clone(), system_clock.clone()).await? { Some(initialized_clone_manifest) if initialized_clone_manifest.db_state().initialized => @@ -122,7 +128,10 @@ async fn create_clone_manifest + Clone>( ) .await?; } - return Ok(initialized_clone_manifest); + return Ok(CreateCloneManifestResult { + clone_manifest: initialized_clone_manifest, + copy_wal_params: None, + }); } Some(uninitialized_clone_manifest) => { for source_spec in &source_specs { @@ -132,7 +141,24 @@ async fn create_clone_manifest + Clone>( &uninitialized_clone_manifest, )?; } - uninitialized_clone_manifest + let copy_wal_params = match &source_specs[..] { + [source_spec] => { + let source = rebuild_source( + source_spec, + &uninitialized_clone_manifest, + &object_store, + &system_clock, + &rand, + &projection_range, + segment_filter.as_ref(), + segment_projection.as_ref(), + ) + .await?; + Some(copy_wal_params_for_source(&source, &clone_path)) + } + _ => None, + }; + (uninitialized_clone_manifest, copy_wal_params) } None => { let sources = build_sources( @@ -145,28 +171,48 @@ async fn create_clone_manifest + Clone>( segment_projection.as_ref(), ) .await?; + let copy_wal_params = match &sources[..] { + [source] => Some(copy_wal_params_for_source(source, &clone_path)), + _ => None, + }; - let manifest: Manifest = match &sources[..] { - // no need to call validate_no_data_wal() because for single source, - // WAL is copied by the caller (create_clone) - [single_source] => Manifest::cloned( - &single_source.manifest, - single_source.path.to_string(), - single_source.checkpoint.id, - rand, - ), + let projection_requested = projection_range.is_some() + || segment_filter.is_some() + || segment_projection.is_some() + || source_specs.iter().any(|s| s.projection_range.is_some()); + + let mut manifest: Manifest = match &sources[..] { + [single_source] => { + // WAL SSTs are copied to the clone verbatim and replayed in full + // when the clone is opened, so entries outside the projected + // range would leak into the clone. So we reject projections if + // there are non-fence WALs to copy. + if projection_requested { + validate_no_data_wal(&sources, wal_admin).await?; + } + Manifest::cloned( + &single_source.manifest, + single_source.path.to_string(), + single_source.checkpoint.id, + rand.clone(), + ) + } [..] => { - validate_no_data_wal(&sources, &wal_object_store).await?; - Manifest::cloned_from_union(sources, rand)? + validate_no_data_wal(&sources, wal_admin).await?; + Manifest::cloned_from_union(sources, rand.clone())? } }; + manifest.core.initialized = false; - StoredManifest::store_uninitialized_clone( - clone_manifest_store, - manifest, - system_clock.clone(), + ( + StoredManifest::store_uninitialized_clone( + clone_manifest_store, + manifest, + system_clock.clone(), + ) + .await?, + copy_wal_params, ) - .await? } }; @@ -212,7 +258,10 @@ async fn create_clone_manifest + Clone>( } } - Ok(clone_manifest) + Ok(CreateCloneManifestResult { + clone_manifest, + copy_wal_params, + }) } fn to_byte_range + Clone>(bounds: &T) -> BytesRange { @@ -226,6 +275,12 @@ pub(crate) struct CloneSource { pub checkpoint: Checkpoint, } +impl CloneSource { + fn versioned_manifest(&self) -> VersionedManifest { + VersionedManifest::from_manifest(self.checkpoint.manifest_id, self.manifest.clone()) + } +} + /// Builds a list of clone sources from the provided specifications. For each source spec, a /// manifest at the specified checkpoint is loaded (if the checkpoint is not specified then it is /// created). Additionally, if any of `projection_range`, `segment_filter`, or @@ -242,40 +297,115 @@ async fn build_sources + Clone>( ) -> Result, SlateDBError> { let mut result: Vec = vec![]; for source in source_specs { - let manifest_store = Arc::new(ManifestStore::new(&source.path, object_store.clone())); - let mut latest_manifest = - load_initialized_manifest(manifest_store.clone(), system_clock.clone()).await?; - let checkpoint = - get_or_create_parent_checkpoint(&mut latest_manifest, source.checkpoint, rand.clone()) - .await?; - let mut manifest_at_checkpoint = - manifest_store.read_manifest(checkpoint.manifest_id).await?; - - let range: Option = match (source.projection_range.clone(), projection_range) { - (Some(l), Some(r)) => to_byte_range(&l).intersect(&to_byte_range(r)), - (Some(l), None) => Some(to_byte_range(&l)), - (None, Some(r)) => Some(to_byte_range(r)), - (None, None) => None, - }; + result.push( + build_source( + source, + source.checkpoint, + object_store, + system_clock, + rand, + projection_range, + segment_filter, + segment_projection, + ) + .await?, + ); + } + Ok(result) +} - let config = ProjectionConfig { - global_range: range, - segment_filter: segment_filter.cloned(), - segment_projection: segment_projection.cloned(), - }; - manifest_at_checkpoint = if config.is_noop() { - manifest_at_checkpoint - } else { - Manifest::projected(&manifest_at_checkpoint, &config)? - }; +async fn build_source + Clone>( + source: &CloneSourceSpec, + checkpoint_id: Option, + object_store: &Arc, + system_clock: &Arc, + rand: &Arc, + projection_range: &Option, + segment_filter: Option<&SegmentFilterFn>, + segment_projection: Option<&SegmentProjectionFn>, +) -> Result { + let manifest_store = Arc::new(ManifestStore::new(&source.path, object_store.clone())); + let mut latest_manifest = + load_initialized_manifest(manifest_store.clone(), system_clock.clone()).await?; + let checkpoint = + get_or_create_parent_checkpoint(&mut latest_manifest, checkpoint_id, rand.clone()).await?; + let mut manifest_at_checkpoint = manifest_store.read_manifest(checkpoint.manifest_id).await?; + + let range: Option = match (source.projection_range.clone(), projection_range) { + (Some(l), Some(r)) => to_byte_range(&l).intersect(&to_byte_range(r)), + (Some(l), None) => Some(to_byte_range(&l)), + (None, Some(r)) => Some(to_byte_range(r)), + (None, None) => None, + }; - result.push(CloneSource { - path: source.path.clone(), - manifest: manifest_at_checkpoint, - checkpoint, - }); + let config = ProjectionConfig { + global_range: range, + segment_filter: segment_filter.cloned(), + segment_projection: segment_projection.cloned(), + }; + manifest_at_checkpoint = if config.is_noop() { + manifest_at_checkpoint + } else { + Manifest::projected(&manifest_at_checkpoint, &config)? + }; + + Ok(CloneSource { + path: source.path.clone(), + manifest: manifest_at_checkpoint, + checkpoint, + }) +} + +fn copy_wal_params_for_source(source: &CloneSource, to_path: &Path) -> CopyWalParams { + CopyWalParams { + from_path: source.path.clone(), + from_manifest: source.versioned_manifest(), + to_path: to_path.clone(), } - Ok(result) +} + +async fn rebuild_source + Clone>( + source_spec: &CloneSourceSpec, + clone_manifest: &StoredManifest, + object_store: &Arc, + system_clock: &Arc, + rand: &Arc, + projection_range: &Option, + segment_filter: Option<&SegmentFilterFn>, + segment_projection: Option<&SegmentProjectionFn>, +) -> Result { + // `Manifest::cloned` appends the direct parent after inherited external DBs. Search in reverse + // so a parent that also appears in its own ancestry still resolves to the direct source. + let source_path = source_spec.path.to_string(); + let external_db = clone_manifest + .manifest() + .external_dbs + .iter() + .rev() + .find(|external_db| external_db.path == source_path) + .ok_or(SlateDBError::CloneExternalDbMissing)?; + let manifest_store = Arc::new(ManifestStore::new(&source_spec.path, object_store.clone())); + let latest_manifest = load_initialized_manifest(manifest_store, system_clock.clone()).await?; + let checkpoint_id = external_db + .final_checkpoint_id + .filter(|checkpoint_id| { + latest_manifest + .db_state() + .find_checkpoint(*checkpoint_id) + .is_some() + }) + .unwrap_or(external_db.source_checkpoint_id); + build_source( + source_spec, + Some(checkpoint_id), + object_store, + system_clock, + rand, + projection_range, + segment_filter, + segment_projection, + ) + .await } // Get a checkpoint and the corresponding manifest that will be used as the source @@ -312,16 +442,16 @@ async fn get_or_create_parent_checkpoint( } fn validate_clone_source_specs + Clone>( - specs: Vec>, - clone_path: Path, + specs: &[CloneSourceSpec], + clone_path: &Path, ) -> Result<(), SlateDBError> { if specs.is_empty() { return Err(SlateDBError::InvalidUnionSetEmpty()); } let mut seen_paths = std::collections::HashSet::new(); - for source in &specs { - if clone_path == source.path { + for source in specs { + if clone_path == &source.path { return Err(SlateDBError::IdenticalClonePaths(clone_path.clone())); } if !seen_paths.insert(source.path.to_string()) { @@ -333,42 +463,26 @@ fn validate_clone_source_specs + Clone>( async fn validate_no_data_wal( sources: &[CloneSource], - wal_object_store: &Arc, + wal_admin: &dyn WalAdmin, ) -> Result<(), SlateDBError> { let mut parents_with_wal = vec![]; for source in sources { - let core = &source.manifest.core; - // Cheap manifest check first: if the WAL range is empty there is - // nothing to inspect. - if core.next_wal_sst_id.saturating_sub(1) <= core.replay_after_wal_id { - continue; - } - - let path_resolver = PathResolver::from_root(source.path.clone()); - let mut has_data_wal = false; - for wal_id in (core.replay_after_wal_id + 1)..core.next_wal_sst_id { - let path = path_resolver.sst_path(&SsTableId::Wal(wal_id)); - match wal_object_store.head(&path).await { - Ok(meta) => { - // Fence WALs are zero-byte `SsTableId::Wal` objects (written via - // `TableStore::write_wal_fence`). They contain no data, so dropping them loses - // nothing and the source is allowed to participate in the union. We use the - // same zero-size discriminator that fence garbage_collector/wal_gc.rs uses. - if meta.size > 0 { - has_data_wal = true; - break; - } - } - Err(e) => return Err(SlateDBError::from(e)), - } - } - - if has_data_wal { + let replay_after_wal_id = source.manifest.core.replay_after_wal_id; + let wal_id_last_seen = source + .manifest + .core + .next_wal_sst_id + .checked_sub(1) + .ok_or(SlateDBError::InvalidDBState)?; + if !wal_admin + .is_empty(&source.path, replay_after_wal_id, wal_id_last_seen) + .await? + { parents_with_wal.push(source.path.clone()); } } if !parents_with_wal.is_empty() { - return Err(SlateDBError::InvalidUnionSourceWithWal { + return Err(SlateDBError::InvalidCloneSourceWithWal { paths: parents_with_wal, }); } @@ -455,36 +569,24 @@ async fn load_initialized_manifest( Ok(manifest) } -async fn copy_wal_ssts( - object_store: Arc, - parent_checkpoint_state: &ManifestCore, - parent_path: &Path, - clone_path: &Path, - #[allow(unused)] fp_registry: Arc, -) -> Result<(), SlateDBError> { - let parent_path_resolver = PathResolver::from_root(parent_path.clone()); - let clone_path_resolver = PathResolver::from_root(clone_path.clone()); - - let mut wal_id = parent_checkpoint_state.replay_after_wal_id + 1; - while wal_id < parent_checkpoint_state.next_wal_sst_id { - fail_point!(fp_registry.clone(), "copy-wal-ssts-io-error", |_| Err( - SlateDBError::from(std::io::Error::other("oops")) - )); - - let id = SsTableId::Wal(wal_id); - let parent_path = parent_path_resolver.sst_path(&id); - let clone_path = clone_path_resolver.sst_path(&id); - object_store - .as_ref() - .copy(&parent_path, &clone_path) - .await?; - wal_id += 1; - } - Ok(()) +async fn copy_wal( + wal_admin: &dyn WalAdmin, + params: CopyWalParams, +) -> Result<(u64, u64), SlateDBError> { + let CopyWalParams { + from_path, + from_manifest, + to_path, + } = params; + wal_admin + .clone_wal(&from_path, from_manifest, &to_path) + .await + .map_err(Into::into) } #[cfg(test)] mod tests { + use super::{SegmentFilterFn, SegmentProjectionFn}; use crate::config::{ CheckpointOptions, CheckpointScope, FlushOptions, FlushType, PutOptions, Settings, WriteOptions, @@ -497,12 +599,15 @@ mod tests { use crate::iter::IterationOrder; use crate::manifest::store::{ManifestStore, StoredManifest}; use crate::manifest::Manifest; - use crate::manifest::ManifestCore; + use crate::manifest::{ManifestCore, VersionedManifest}; use crate::object_stores::ObjectStores; use crate::paths::PathResolver; use crate::proptest_util::{rng, sample}; use crate::test_utils; use crate::utils::IdGenerator; + use crate::wal::slatedb::admin::SlateDbWalAdmin; + use crate::wal::{WalAdmin, WalError, WalFileRange, WalGc}; + use async_trait::async_trait; use bytes::Bytes; use fail_parallel::FailPointRegistry; use object_store::memory::InMemory; @@ -517,8 +622,94 @@ mod tests { use std::ops::Bound; use std::ops::RangeBounds; use std::sync::Arc; + use std::time::Duration; use uuid::Uuid; + struct RemappingWalAdmin { + replay_range: (u64, u64), + expected_manifest_id: Option, + } + + struct NoopWalGc; + + #[async_trait] + impl WalGc for NoopWalGc { + async fn collect( + &self, + _referenced_ranges: Vec, + _min_age: Duration, + _dry_run: bool, + ) -> Result<(), WalError> { + Ok(()) + } + } + + #[async_trait] + impl WalAdmin for RemappingWalAdmin { + fn garbage_collector(&self, _path: &Path) -> Arc { + Arc::new(NoopWalGc) + } + + async fn delete_wal(&self, _path: &Path, _dry_run: bool) -> Result, WalError> { + Ok(vec![]) + } + + async fn is_empty( + &self, + _path: &Path, + _replay_after_wal_id: u64, + _wal_id_last_seen: u64, + ) -> Result { + Ok(true) + } + + async fn clone_wal( + &self, + _from_path: &Path, + from_manifest: VersionedManifest, + _to_path: &Path, + ) -> Result<(u64, u64), WalError> { + if let Some(expected_manifest_id) = self.expected_manifest_id { + assert_eq!(from_manifest.id(), expected_manifest_id); + } + Ok(self.replay_range) + } + } + + async fn create_native_clone, R: RangeBounds + Clone>( + clone_sources: Vec>, + clone_path: P, + object_stores: ObjectStores, + fp_registry: Arc, + system_clock: Arc, + rand: Arc, + projection_range: Option, + segment_filter: Option, + segment_projection: Option, + ) -> Result<(), SlateDBError> { + let wal_admin = Arc::new(SlateDbWalAdmin::new( + object_stores + .store_of(crate::object_stores::ObjectStoreType::Wal) + .clone(), + fp_registry.clone(), + )); + crate::clone::create_clone( + clone_sources, + clone_path, + object_stores + .store_of(crate::object_stores::ObjectStoreType::Main) + .clone(), + wal_admin, + fp_registry, + system_clock, + rand, + projection_range, + segment_filter, + segment_projection, + ) + .await + } + // helper method for tests that creates CloneSourceSpec async fn create_clone>( clone_path: P, @@ -534,7 +725,7 @@ mod tests { Some(cp) => CloneSourceSpec::with_checkpoint(parent_path, cp), None => CloneSourceSpec::new(parent_path), }; - crate::clone::create_clone( + create_native_clone( vec![source], clone_path, ObjectStores::new(object_store, Some(wal_object_store)), @@ -548,6 +739,105 @@ mod tests { .await } + #[tokio::test] + async fn should_stamp_wal_range_returned_by_wal_admin() { + let object_store: Arc = Arc::new(InMemory::new()); + let parent_path = Path::from("/tmp/test_parent_remapped_wal"); + let clone_path = Path::from("/tmp/test_clone_remapped_wal"); + let system_clock: Arc = Arc::new(DefaultSystemClock::new()); + + let mut parent_manifest = StoredManifest::create_new_db( + Arc::new(ManifestStore::new(&parent_path, object_store.clone())), + ManifestCore::new(), + system_clock.clone(), + ) + .await + .unwrap(); + let checkpoint = parent_manifest + .write_checkpoint(Uuid::new_v4(), &CheckpointOptions::default()) + .await + .unwrap(); + + let wal_admin = RemappingWalAdmin { + replay_range: (41, 46), + expected_manifest_id: Some(checkpoint.manifest_id), + }; + let source: CloneSourceSpec = CloneSourceSpec::with_checkpoint(parent_path, checkpoint.id); + crate::clone::create_clone( + vec![source], + clone_path.clone(), + object_store.clone(), + Arc::new(wal_admin), + Arc::new(FailPointRegistry::new()), + system_clock.clone(), + Arc::new(DbRand::default()), + None, + None, + None, + ) + .await + .unwrap(); + + let manifest = StoredManifest::load( + Arc::new(ManifestStore::new(&clone_path, object_store)), + system_clock, + ) + .await + .unwrap(); + assert!(manifest.db_state().initialized); + assert_eq!(manifest.db_state().replay_after_wal_id, 41); + assert_eq!(manifest.db_state().next_wal_sst_id, 47); + } + + #[tokio::test] + async fn should_reset_wal_range_when_clone_does_not_copy_wal() { + let object_store: Arc = Arc::new(InMemory::new()); + let parent_paths = [ + Path::from("/tmp/test_parent_no_wal_a"), + Path::from("/tmp/test_parent_no_wal_b"), + ]; + let clone_path = Path::from("/tmp/test_clone_no_wal"); + let system_clock: Arc = Arc::new(DefaultSystemClock::new()); + + for parent_path in &parent_paths { + StoredManifest::create_new_db( + Arc::new(ManifestStore::new(parent_path, object_store.clone())), + ManifestCore::new(), + system_clock.clone(), + ) + .await + .unwrap(); + } + + crate::clone::create_clone( + parent_paths.into_iter().map(CloneSourceSpec::new).collect(), + clone_path.clone(), + object_store.clone(), + Arc::new(RemappingWalAdmin { + replay_range: (41, 47), + expected_manifest_id: None, + }), + Arc::new(FailPointRegistry::new()), + system_clock.clone(), + Arc::new(DbRand::default()), + None, + None, + None, + ) + .await + .unwrap(); + + let manifest = StoredManifest::load( + Arc::new(ManifestStore::new(&clone_path, object_store)), + system_clock, + ) + .await + .unwrap(); + assert!(manifest.db_state().initialized); + assert_eq!(manifest.db_state().replay_after_wal_id, 0); + assert_eq!(manifest.db_state().next_wal_sst_id, 1); + } + #[tokio::test] async fn should_clone_latest_state_if_no_checkpoint_provided() { let mut rng = rng::new_test_rng(None); @@ -822,7 +1112,7 @@ mod tests { // Create an uninitialized manifest with an invalid checkpoint id let clone_manifest_store = Arc::new(ManifestStore::new(&clone_path, object_store.clone())); - let non_existent_source_checkpoint_id = uuid::Uuid::new_v4(); + let non_existent_source_checkpoint_id = Uuid::new_v4(); StoredManifest::store_uninitialized_clone( clone_manifest_store, Manifest::cloned( @@ -936,7 +1226,7 @@ mod tests { Manifest::cloned( &parent_manifest, original_parent_path.to_string(), - uuid::Uuid::new_v4(), + Uuid::new_v4(), rand.clone(), ), system_clock.clone(), @@ -1085,7 +1375,7 @@ mod tests { ) .await .unwrap_err(); - assert!(matches!(err, SlateDBError::IoError(_))); + assert!(matches!(err, SlateDBError::WalUnavailable(_))); fail_parallel::cfg(Arc::clone(&fp_registry), "copy-wal-ssts-io-error", "off").unwrap(); create_clone( @@ -1348,14 +1638,167 @@ mod tests { .unwrap_err(); assert!(matches!( err, - SlateDBError::ObjectStoreError(ref source) + SlateDBError::WalUnavailable(ref source) if matches!( - source.as_ref(), - ObjectStoreError::NotFound { path, .. } if path == &expected_missing_wal_path + source.downcast_ref::(), + Some(ObjectStoreError::NotFound { path, .. }) + if path == &expected_missing_wal_path ) )); } + #[tokio::test] + async fn should_disallow_projected_clone_when_source_has_data_wal() { + // Data that only lives in the parent's WAL at the checkpoint is copied + // to the clone verbatim and replayed in full on first open, so a + // projection cannot be applied to it. Cloning with a projection must + // fail while the source still has data in its WAL, and succeed once + // that data has been flushed into L0. + let fp_registry = Arc::new(FailPointRegistry::new()); + let object_store: Arc = Arc::new(InMemory::new()); + let parent_path = Path::from("/tmp/test_parent_wal_projection"); + let clone_path = Path::from("/tmp/test_clone_wal_projection"); + + let parent_db = Db::builder(parent_path.clone(), object_store.clone()) + .with_fp_registry(fp_registry.clone()) + .build() + .await + .unwrap(); + let write_options = WriteOptions::default(); + let put_options = PutOptions::default(); + + // Keys inside and outside the projection range [aaa, bbb), flushed + // through to L0 ... + parent_db + .put_with_options(b"aaa-l0", b"v1", &put_options, &write_options) + .await + .unwrap(); + parent_db + .put_with_options(b"zzz-l0", b"v2", &put_options, &write_options) + .await + .unwrap(); + parent_db.flush().await.unwrap(); + parent_db + .flush_with_options(FlushOptions { + flush_type: FlushType::MemTable, + }) + .await + .unwrap(); + + // ... and the same shape of data made durable only in the WAL. + parent_db + .put_with_options(b"aaa-wal", b"v3", &put_options, &write_options) + .await + .unwrap(); + parent_db + .put_with_options(b"zzz-wal", b"v4", &put_options, &write_options) + .await + .unwrap(); + parent_db.flush().await.unwrap(); + + let manifest = parent_db.manifest(); + assert!( + !manifest.manifest.core.tree.l0.is_empty(), + "expected parent state to include L0 data" + ); + assert!( + manifest.manifest.core.replay_after_wal_id + 1 < manifest.manifest.core.next_wal_sst_id, + "expected parent state to retain WAL-only SSTs" + ); + + // Block L0 uploads so the WAL-only data stays in the WAL. + fail_parallel::cfg( + fp_registry.clone(), + "write-compacted-sst-io-error", + "return", + ) + .unwrap(); + // expect to fail since l0 upload is blocked + assert!(parent_db.close().await.is_err()); + fail_parallel::cfg(fp_registry.clone(), "write-compacted-sst-io-error", "off").unwrap(); + + // Cloning with a projection that keeps only keys in [aaa, bbb) must + // be rejected while the WAL-only data is still in the WAL. + let range = ( + Bound::Included(Bytes::from_static(b"aaa")), + Bound::Excluded(Bytes::from_static(b"bbb")), + ); + let err = create_native_clone( + vec![CloneSourceSpec::new(parent_path.clone())], + clone_path.clone(), + ObjectStores::new(object_store.clone(), Some(object_store.clone())), + Arc::new(FailPointRegistry::new()), + Arc::new(DefaultSystemClock::new()), + Arc::new(DbRand::default()), + Some(range.clone()), + None, + None, + ) + .await + .unwrap_err(); + assert!(matches!( + err, + SlateDBError::InvalidCloneSourceWithWal { ref paths } + if paths == &vec![parent_path.clone()] + )); + + // Reopen the parent so the WAL tail is replayed, flush it into L0, + // and close cleanly. With no data WALs left to copy the projected + // clone is allowed. + let parent_db = Db::open(parent_path.clone(), object_store.clone()) + .await + .unwrap(); + parent_db + .flush_with_options(FlushOptions { + flush_type: FlushType::MemTable, + }) + .await + .unwrap(); + parent_db.close().await.unwrap(); + + create_native_clone( + vec![CloneSourceSpec::new(parent_path.clone())], + clone_path.clone(), + ObjectStores::new(object_store.clone(), Some(object_store.clone())), + Arc::new(FailPointRegistry::new()), + Arc::new(DefaultSystemClock::new()), + Arc::new(DbRand::default()), + Some(range), + None, + None, + ) + .await + .unwrap(); + + let clone_db = Db::open(clone_path.clone(), object_store.clone()) + .await + .unwrap(); + + // L0 data respects the projection. + assert_eq!( + clone_db.get(b"aaa-l0").await.unwrap(), + Some(Bytes::from_static(b"v1")) + ); + assert_eq!( + clone_db.get(b"zzz-l0").await.unwrap(), + None, + "L0 entry outside the projection range must not be visible in the clone" + ); + + // The formerly WAL-only data was flushed into L0 before the retry, + // so it must respect the projection too. + assert_eq!( + clone_db.get(b"aaa-wal").await.unwrap(), + Some(Bytes::from_static(b"v3")) + ); + assert_eq!( + clone_db.get(b"zzz-wal").await.unwrap(), + None, + "entry outside the projection range must not be visible in the clone" + ); + clone_db.close().await.unwrap(); + } + fn segmented_table() -> BTreeMap { BTreeMap::from([ (Bytes::from_static(b"aaa-001"), Bytes::from_static(b"v1")), @@ -1382,15 +1825,25 @@ mod tests { settings: Settings, table: &BTreeMap, ) { + #[cfg(feature = "wal_disable")] + let wal_enabled = settings.wal_enabled; + #[cfg(not(feature = "wal_disable"))] + let wal_enabled = true; let db = Db::builder(path.clone(), object_store) .with_settings(settings) .with_segment_extractor(extractor) .build() .await .unwrap(); - // await_durable would deadlock under wal_enabled=false because the + // Do not await the returned handle here: with wal_enabled=false, the // memtable flush is gated on the explicit call below. test_utils::seed_database(&db, table, false).await.unwrap(); + if wal_enabled { + // Flush the WAL before the memtable so that `replay_after_wal_id` + // covers every data WAL; projected clones of this parent would + // otherwise be rejected. + db.flush().await.unwrap(); + } db.flush_with_options(FlushOptions { flush_type: FlushType::MemTable, }) @@ -1419,7 +1872,7 @@ mod tests { object_store: Arc, projection: Option, ) { - crate::clone::create_clone( + create_native_clone( sources, clone_path.clone(), ObjectStores::new(object_store.clone(), Some(object_store)), @@ -1686,6 +2139,112 @@ mod tests { clone_db.close().await.unwrap(); } + #[cfg(feature = "wal_disable")] + #[tokio::test] + async fn should_union_segmented_shards_that_each_span_every_segment() { + // Rescale-down of a store keyed `data/{tenant}/…` and + // `idx/{tenant}/…`, sharded by tenant. Each shard holds part of both + // segments, so the shards' overall key ranges overlap — + // `data/metro…` sorts below `idx/bronx…` — while neither segment + // does. One union call must merge them; no per-source projection and + // no staged re-slicing clones are needed. + let object_store: Arc = Arc::new(InMemory::new()); + let parent_path_a = Path::from("/tmp/test_parent_seg_interleaved_a"); + let parent_path_b = Path::from("/tmp/test_parent_seg_interleaved_b"); + let clone_path = Path::from("/tmp/test_clone_seg_interleaved"); + let extractor = Arc::new(test_utils::DataIdxPrefixExtractor); + let settings = wal_disabled_settings(); + + fn shard(tenants: [&str; 2]) -> BTreeMap { + let mut table = BTreeMap::new(); + for tenant in tenants { + table.insert( + Bytes::from(format!("data/{}/animal/lion-1", tenant)), + Bytes::from(format!("{} lion", tenant)), + ); + table.insert( + Bytes::from(format!("idx/{}/owner/alice/lion-1", tenant)), + Bytes::new(), + ); + } + table + } + let table_a = shard(["bronx", "lincoln"]); + let table_b = shard(["metro", "oakland"]); + + build_segmented_parent( + &parent_path_a, + object_store.clone(), + extractor.clone(), + settings.clone(), + &table_a, + ) + .await; + build_segmented_parent( + &parent_path_b, + object_store.clone(), + extractor.clone(), + settings.clone(), + &table_b, + ) + .await; + + run_segmented_clone( + vec![ + CloneSourceSpec::new(parent_path_a.clone()), + CloneSourceSpec::new(parent_path_b.clone()), + ], + &clone_path, + object_store.clone(), + None, + ) + .await; + + // Both shards contribute an L0 SST to each of the two segments. + let store = ManifestStore::new(&clone_path, object_store.clone()); + let stored = store.read_latest_manifest().await.unwrap(); + assert_eq!( + stored.manifest.core.segment_extractor_name.as_deref(), + Some("data-idx") + ); + let segments: Vec<(Bytes, usize)> = stored + .manifest + .core + .segments + .iter() + .map(|s| (s.prefix.clone(), s.tree.l0.len())) + .collect(); + assert_eq!( + segments, + vec![ + (Bytes::from_static(b"data"), 2), + (Bytes::from_static(b"idx"), 2) + ] + ); + assert_eq!(stored.manifest.external_dbs.len(), 2); + + let mut expected: BTreeMap = table_a.clone(); + expected.extend(table_b.clone()); + + let clone_db = + open_segmented_clone(&clone_path, object_store.clone(), extractor, settings).await; + let mut full_iter = clone_db.scan(..).await.unwrap(); + test_utils::assert_ranged_db_scan(&expected, .., IterationOrder::Ascending, &mut full_iter) + .await; + // Each segment routes reads across both shards' contributions. + assert_segment_prefix_scan(&clone_db, &expected, b"data", b"datb").await; + assert_segment_prefix_scan(&clone_db, &expected, b"idx", b"idy").await; + for (key, value) in &expected { + assert_eq!( + clone_db.get(key).await.unwrap().as_ref(), + Some(value), + "key={:?}", + key + ); + } + clone_db.close().await.unwrap(); + } + #[cfg(feature = "wal_disable")] #[tokio::test] async fn should_union_projected_segmented_dbs() { @@ -2063,7 +2622,7 @@ mod tests { // Source B has no extra WAL. build_plain_wal_disabled_parent(&parent_path_b, object_store.clone(), &table_b).await; - crate::clone::create_clone( + create_native_clone( vec![ CloneSourceSpec::new(parent_path_a.clone()), CloneSourceSpec::new(parent_path_b.clone()), @@ -2095,7 +2654,7 @@ mod tests { } /// A union clone whose source references a real (non-empty) data WAL above - /// `replay_after_wal_id` must FAIL with `InvalidUnionSourceWithWal`, since + /// `replay_after_wal_id` must FAIL with `InvalidCloneSourceWithWal`, since /// the union clone would silently drop that WAL data. #[cfg(feature = "wal_disable")] #[tokio::test] @@ -2126,7 +2685,7 @@ mod tests { .await; build_plain_wal_disabled_parent(&parent_path_b, object_store.clone(), &table_b).await; - let err = crate::clone::create_clone( + let err = create_native_clone( vec![ CloneSourceSpec::new(parent_path_a.clone()), CloneSourceSpec::new(parent_path_b.clone()), @@ -2144,10 +2703,10 @@ mod tests { .unwrap_err(); match err { - SlateDBError::InvalidUnionSourceWithWal { paths } => { + SlateDBError::InvalidCloneSourceWithWal { paths } => { assert!(paths.contains(&parent_path_a)); } - other => panic!("expected InvalidUnionSourceWithWal, got {other:?}"), + other => panic!("expected InvalidCloneSourceWithWal, got {other:?}"), } } @@ -2195,7 +2754,7 @@ mod tests { })) .to_string(); - let err = crate::clone::create_clone( + let err = create_native_clone( vec![ CloneSourceSpec::new(parent_path_a.clone()), CloneSourceSpec::new(parent_path_b.clone()), @@ -2215,10 +2774,10 @@ mod tests { assert!( matches!( err, - SlateDBError::ObjectStoreError(ref source) + SlateDBError::WalUnavailable(ref source) if matches!( - source.as_ref(), - ObjectStoreError::NotFound { path, .. } + source.downcast_ref::(), + Some(ObjectStoreError::NotFound { path, .. }) if path == &expected_missing_wal_path ) ), diff --git a/slatedb/src/compaction_execute_bench.rs b/slatedb/src/compaction_execute_bench.rs index d5e145bb88..c7d4c71603 100644 --- a/slatedb/src/compaction_execute_bench.rs +++ b/slatedb/src/compaction_execute_bench.rs @@ -1,5 +1,4 @@ use std::collections::HashMap; -use std::mem; use std::sync::Arc; use std::time::Duration; @@ -83,7 +82,7 @@ impl CompactionExecuteBench { BlockCachePolicy::default(), )); let num_keys = sst_bytes / (val_bytes + key_bytes); - let mut key_start = vec![0u8; key_bytes - mem::size_of::()]; + let mut key_start = vec![0u8; key_bytes - size_of::()]; self.rand.rng().fill_bytes(key_start.as_mut_slice()); let mut futures = FuturesUnordered::>>::new(); for i in 0..num_ssts { @@ -183,7 +182,7 @@ impl CompactionExecuteBench { .now() .signed_duration_since(start) .num_milliseconds(); - info!("wrote sst [id={:?}, elapsed_ms={}]", &sst.id, elapsed_ms); + info!("wrote sst [id={:?}, elapsed_ms={}]", sst.id, elapsed_ms); Ok(()) } diff --git a/slatedb/src/compaction_worker.rs b/slatedb/src/compaction_worker.rs index 3b58f3790d..836e0c4aa6 100644 --- a/slatedb/src/compaction_worker.rs +++ b/slatedb/src/compaction_worker.rs @@ -616,7 +616,13 @@ impl CompactionWorkerHandler { let heartbeat_ms = self.clock.now().timestamp_millis() as u64; let updated = existing .with_status(CompactionStatus::Compacted) - .with_output_ssts(sorted_run.sst_views.iter().map(|v| v.sst.clone()).collect()) + .with_output_ssts( + sorted_run + .sst_views() + .iter() + .map(|v| v.sst.clone()) + .collect(), + ) .with_worker(Some(WorkerSpec::new(self.worker_id.clone(), heartbeat_ms))) .with_ctx(None); dirty.value.insert(updated); @@ -1114,10 +1120,7 @@ mod tests { ..SsTableInfo::default() }, ); - let sorted_run = SortedRun { - id: 0, - sst_views: vec![SsTableView::identity(output_handle.clone())], - }; + let sorted_run = SortedRun::new(0, [SsTableView::identity(output_handle.clone())]); fx.handler .handle_finished(id, Ok(sorted_run)) @@ -1373,13 +1376,13 @@ mod tests { let (tx, rx) = async_channel::unbounded::(); let executor: Arc = Arc::new( TokioCompactionExecutor::new(TokioCompactionExecutorOptions { - handle: tokio::runtime::Handle::current(), + handle: Handle::current(), options: options.clone(), worker_tx: tx, table_store: table_store.clone(), rand: Arc::new(DbRand::new(100u64)), stats: { - let recorder = slatedb_common::metrics::MetricsRecorderHelper::noop(); + let recorder = MetricsRecorderHelper::noop(); Arc::new(CompactionStats::new(&recorder)) }, worker_stats: WorkerStats::noop(), @@ -1425,7 +1428,7 @@ mod tests { COMPACTION_WORKER_TASK_NAME.to_string(), Box::new(handler), rx, - &tokio::runtime::Handle::current(), + &Handle::current(), ) .unwrap(); let worker = CompactionWorker::new(task_executor); diff --git a/slatedb/src/compactions_store.rs b/slatedb/src/compactions_store.rs index e264b244a7..45e4c37361 100644 --- a/slatedb/src/compactions_store.rs +++ b/slatedb/src/compactions_store.rs @@ -276,7 +276,6 @@ impl CompactionsStore { mod tests { use super::*; use crate::compactor_state::{Compaction, CompactionSpec, SourceId}; - use crate::error; use crate::retrying_object_store::RetryingObjectStore; use crate::test_utils::FlakyObjectStore; use object_store::memory::InMemory; @@ -301,7 +300,7 @@ mod tests { assert!(matches!( result.unwrap_err(), - error::SlateDBError::TransactionalObjectVersionExists + SlateDBError::TransactionalObjectVersionExists )); } @@ -395,7 +394,7 @@ mod tests { .unwrap(); let result = compactor1.refresh().await; - assert!(matches!(result, Err(error::SlateDBError::Fenced))); + assert!(matches!(result, Err(SlateDBError::Fenced))); } #[tokio::test] diff --git a/slatedb/src/compactor.rs b/slatedb/src/compactor.rs index 39a65331f9..baa4861847 100644 --- a/slatedb/src/compactor.rs +++ b/slatedb/src/compactor.rs @@ -188,7 +188,7 @@ pub trait CompactionScheduler: Send + Sync { "rejected full-segment compaction: unknown segment {:?}", segment ); - return Err(crate::Error::from(SlateDBError::InvalidCompaction)); + return Err(Error::from(SlateDBError::InvalidCompaction)); }; match plan_full_tree(segment, tree) { Some(spec) => Ok(vec![spec]), @@ -200,7 +200,7 @@ pub trait CompactionScheduler: Send + Sync { segment ); } - Err(crate::Error::from(SlateDBError::InvalidCompaction)) + Err(Error::from(SlateDBError::InvalidCompaction)) } } } @@ -373,12 +373,12 @@ impl Compactor { /// /// ## Returns /// - `Ok(())` when the compactor task exits cleanly, or [`SlateDBError`] on failure. - pub async fn run(&self) -> Result<(), crate::Error> { + pub async fn run(&self) -> Result<(), Error> { self.start().await?; self.join().await } - pub(crate) async fn start(&self) -> Result<(), crate::Error> { + pub(crate) async fn start(&self) -> Result<(), Error> { // The coordinator delegates compaction execution to [`crate::compaction_worker::CompactionWorker`] // either spawned in this process (set `worker: Some`) or running standalone (set `worker: None`). let (_tx, rx) = async_channel::unbounded::(); @@ -401,7 +401,7 @@ impl Compactor { rx, &Handle::current(), ) - .map_err(crate::Error::from)?; + .map_err(Error::from)?; // Spawn an in-process worker if configured. The worker runs under its // own cancellation token; Compactor::stop and run() are responsible for @@ -429,23 +429,23 @@ impl Compactor { worker_rx, &Handle::current(), ) - .map_err(crate::Error::from)?; + .map_err(Error::from)?; } self.task_executor.monitor_on(&Handle::current())?; Ok(()) } - pub(crate) async fn join(&self) -> Result<(), crate::Error> { + pub(crate) async fn join(&self) -> Result<(), Error> { self.task_executor .join_task(COMPACTOR_TASK_NAME) .await - .map_err(crate::Error::from)?; + .map_err(Error::from)?; if self.options.worker.is_some() { self.task_executor .join_task(crate::compaction_worker::COMPACTION_WORKER_TASK_NAME) .await - .map_err(crate::Error::from)?; + .map_err(Error::from)?; } Ok(()) } @@ -454,16 +454,16 @@ impl Compactor { /// /// ## Returns /// - `Ok(())` once the task has shut down, or [`SlateDBError`] if shutdown fails. - pub async fn stop(&self) -> Result<(), crate::Error> { + pub async fn stop(&self) -> Result<(), Error> { self.task_executor .shutdown_task(COMPACTOR_TASK_NAME) .await - .map_err(crate::Error::from)?; + .map_err(Error::from)?; if self.options.worker.is_some() { self.task_executor .shutdown_task(crate::compaction_worker::COMPACTION_WORKER_TASK_NAME) .await - .map_err(crate::Error::from)?; + .map_err(Error::from)?; } Ok(()) } @@ -480,13 +480,13 @@ impl Compactor { compactions_store: Arc, rand: Arc, system_clock: Arc, - ) -> Result { + ) -> Result { let compaction_id = rand.rng().gen_ulid(system_clock.as_ref()); let compaction = Compaction::new(compaction_id, spec); let mut stored_compactions = match StoredCompactions::try_load(compactions_store.clone()).await? { Some(stored) => stored, - None => return Err(crate::Error::from(SlateDBError::InvalidDBState)), + None => return Err(Error::from(SlateDBError::InvalidDBState)), }; loop { @@ -497,7 +497,7 @@ impl Compactor { Err(err) if err.is_sequenced_write_conflict() => { stored_compactions.refresh().await?; } - Err(err) => return Err(crate::Error::from(err)), + Err(err) => return Err(Error::from(err)), } } } @@ -635,8 +635,18 @@ impl CompactorEventHandler { .active_compactions() .filter(|c| c.status() != CompactionStatus::Compacted) { - let estimated_source_bytes = - Self::calculate_estimated_source_bytes(compaction, db_state); + let Some(estimated_source_bytes) = + Self::calculate_estimated_source_bytes(compaction, db_state) + else { + warn!( + "skipping compaction progress because a source is absent from the manifest \ + [id={}, status={:?}, spec={}]", + compaction.id(), + compaction.status(), + compaction.spec(), + ); + continue; + }; total_estimated_bytes += estimated_source_bytes; total_bytes_processed += compaction.bytes_processed(); @@ -661,11 +671,11 @@ impl CompactorEventHandler { 0.0 }; - let percentage = if estimated_source_bytes > 0 { - (compaction.bytes_processed() * 100 / estimated_source_bytes) as u32 - } else { - 0 - }; + let percentage = compaction + .bytes_processed() + .saturating_mul(100) + .checked_div(estimated_source_bytes) + .unwrap_or_default() as u32; debug!( "compaction progress [id={}, progress={}%, processed_bytes={}, estimated_source_bytes={}, elapsed={:.2}s, throughput={}/s]", compaction.id(), @@ -690,10 +700,11 @@ impl CompactorEventHandler { } /// Calculates the estimated total source bytes for a compaction. - fn calculate_estimated_source_bytes(compaction: &Compaction, db_state: &ManifestCore) -> u64 { - let tree = db_state - .tree_for_segment(compaction.spec().segment()) - .expect("compaction target segment missing from manifest"); + fn calculate_estimated_source_bytes( + compaction: &Compaction, + db_state: &ManifestCore, + ) -> Option { + let tree = db_state.tree_for_segment(compaction.spec().segment())?; let views_by_id: HashMap = tree.l0.iter().map(|view| (view.id, view)).collect(); @@ -704,17 +715,13 @@ impl CompactorEventHandler { .spec() .sources() .iter() - .map(|source| match source { - SourceId::SstView(id) => views_by_id - .get(id) - .expect("compaction source view not found in L0") - .estimate_size(), - SourceId::SortedRun(id) => srs_by_id - .get(id) - .expect("compaction source sorted run not found") - .estimate_size(), + .try_fold(0, |total, source| { + let source_bytes = match source { + SourceId::SstView(id) => views_by_id.get(id)?.estimate_size(), + SourceId::SortedRun(id) => srs_by_id.get(id)?.estimate_size(), + }; + Some(total + source_bytes) }) - .sum() } /// Handles a polling tick by refreshing compactions and the manifest, then possibly scheduling compactions. @@ -882,14 +889,13 @@ impl CompactorEventHandler { .spec() .destination() .expect("Compacted tiered compaction must have a destination SR id"); - let output_sr = SortedRun { - id: destination, - sst_views: compaction + let output_sr = SortedRun::new( + destination, + compaction .output_ssts() .iter() - .map(|sst| SsTableView::identity(sst.clone())) - .collect(), - }; + .map(|sst| SsTableView::identity(sst.clone())), + ); self.state_mut().finish_compaction(id, output_sr); manifest_changed = true; self.stats @@ -966,16 +972,12 @@ impl CompactorEventHandler { ); return Err(SlateDBError::InvalidCompaction); }; - let l0_view_ids = tree - .l0 - .iter() - .map(|view| view.id) - .collect::>(); + let l0_view_ids = tree.l0.iter().map(|view| view.id).collect::>(); let sr_ids = tree .compacted .iter() .map(|sr| sr.id) - .collect::>(); + .collect::>(); if let Some(missing) = spec.sources().iter().find(|source| match source { SourceId::SstView(id) => !l0_view_ids.contains(id), @@ -1101,7 +1103,7 @@ impl CompactorEventHandler { if !compaction.is_drain() { return Ok(()); } - let drained_l0_ids: std::collections::HashSet = compaction + let drained_l0_ids: HashSet = compaction .sources() .iter() .filter_map(|s| match s { @@ -1707,7 +1709,7 @@ mod tests { // then: let db_state = db_state.expect("db was not compacted"); for run in db_state.tree.compacted.iter() { - for sst in run.sst_views.iter() { + for sst in run.sst_views() { let mut iter = SstIterator::new_borrowed_initialized( .., sst, @@ -1813,15 +1815,9 @@ mod tests { for key in [b"a", b"b", b"c", b"d"] { batch.put(key, b"value"); } - db.write_with_options( - batch, - &WriteOptions { - await_durable: false, - ..Default::default() - }, - ) - .await - .unwrap(); + db.write_with_options(batch, &WriteOptions::default()) + .await + .unwrap(); db.flush_with_options(FlushOptions { flush_type: FlushType::MemTable, }) @@ -1836,7 +1832,7 @@ mod tests { .tree .compacted .iter() - .flat_map(|sr| &sr.sst_views) + .flat_map(|sr| sr.sst_views().iter()) .collect(); assert_eq!(output_ssts.len(), 1); let view = output_ssts[0]; @@ -2312,8 +2308,8 @@ mod tests { let (manifest_store, _, table_store) = build_test_stores(os.clone()); - // put key 'a' into L1 (and key 'b' so that when we delete 'a' the SST is non-empty) - // since these are both await_durable=true, we're guaranteed to have one L0 SST for each. + // Put key 'a' into L1 (and key 'b' so that when we delete 'a' the SST is non-empty). + // The explicit flush below makes both writes durable. db.put(&[b'a'; 16], &[b'a'; 32]).await.unwrap(); db.put(&[b'b'; 16], &[b'a'; 32]).await.unwrap(); db.flush().await.unwrap(); @@ -2364,7 +2360,7 @@ mod tests { .unwrap(); assert_eq!(db_state.tree.compacted.len(), 1); - let compacted = &db_state.tree.compacted.first().unwrap().sst_views; + let compacted = db_state.tree.compacted.first().unwrap().sst_views(); assert_eq!(compacted.len(), 1); let handle = compacted.first().unwrap(); @@ -2464,7 +2460,7 @@ mod tests { .unwrap(); assert_eq!(db_state.tree.compacted.len(), 1); - let compacted = &db_state.tree.compacted.first().unwrap().sst_views; + let compacted = db_state.tree.compacted.first().unwrap().sst_views(); assert_eq!(compacted.len(), 1); let handle = compacted.first().unwrap(); @@ -2605,7 +2601,7 @@ mod tests { // then: let db_state = db_state.expect("db was not compacted"); assert_eq!(db_state.tree.compacted.len(), 1); - let compacted = &db_state.tree.compacted.first().unwrap().sst_views; + let compacted = db_state.tree.compacted.first().unwrap().sst_views(); assert_eq!(compacted.len(), 1); let handle = compacted.first().unwrap(); @@ -2825,7 +2821,7 @@ mod tests { // then: let db_state = db_state.expect("db was not compacted"); assert_eq!(db_state.tree.compacted.len(), 1); - let compacted = &db_state.tree.compacted.first().unwrap().sst_views; + let compacted = db_state.tree.compacted.first().unwrap().sst_views(); assert_eq!(compacted.len(), 1); let handle = compacted.first().unwrap(); @@ -2947,7 +2943,7 @@ mod tests { // then: let db_state = db_state.expect("db was not compacted"); assert_eq!(db_state.tree.compacted.len(), 1); - let compacted = &db_state.tree.compacted.first().unwrap().sst_views; + let compacted = db_state.tree.compacted.first().unwrap().sst_views(); assert_eq!(compacted.len(), 1); let handle = compacted.first().unwrap(); @@ -3082,7 +3078,7 @@ mod tests { // then: let db_state = db_state.expect("db was not compacted"); assert_eq!(db_state.tree.compacted.len(), 1); - let compacted = &db_state.tree.compacted.first().unwrap().sst_views; + let compacted = db_state.tree.compacted.first().unwrap().sst_views(); assert_eq!(compacted.len(), 1); let handle = compacted.first().unwrap(); @@ -3150,15 +3146,18 @@ mod tests { db.merge_with_options( b"key1", &[b'a'; 32], - &crate::config::MergeOptions { - ttl: Ttl::ExpireAfter(10), + &MergeOptions { + ttl: Ttl::ExpireAfterMillis(10), }, &WriteOptions { - await_durable: true, + await_durable: false, ..Default::default() }, ) .await + .unwrap() + .await_durable() + .await .unwrap(); // ticker time = 20, no expire time @@ -3166,13 +3165,16 @@ mod tests { db.merge_with_options( b"key1", &[b'b'; 32], - &crate::config::MergeOptions { ttl: Ttl::NoExpiry }, + &MergeOptions { ttl: Ttl::NoExpiry }, &WriteOptions { - await_durable: true, + await_durable: false, ..Default::default() }, ) .await + .unwrap() + .await_durable() + .await .unwrap(); let db_state = await_compaction(&db, os.clone(), Some(insert_clock.clone())) @@ -3182,7 +3184,7 @@ mod tests { assert_eq!(db_state.last_l0_clock_tick, 20); // then: the compacted SST should only contain the non-expired logical value - let compacted = &db_state.tree.compacted.first().unwrap().sst_views; + let compacted = db_state.tree.compacted.first().unwrap().sst_views(); assert_eq!(compacted.len(), 1); let handle = compacted.first().unwrap(); @@ -3343,8 +3345,8 @@ mod tests { db.merge_with_options( b"key1", b"a", - &crate::config::MergeOptions { - ttl: Ttl::ExpireAfter(100), + &MergeOptions { + ttl: Ttl::ExpireAfterMillis(100), }, &WriteOptions { await_durable: false, @@ -3369,8 +3371,8 @@ mod tests { db.merge_with_options( b"key1", b"b", - &crate::config::MergeOptions { - ttl: Ttl::ExpireAfter(200), + &MergeOptions { + ttl: Ttl::ExpireAfterMillis(200), }, &WriteOptions { await_durable: false, @@ -3411,7 +3413,7 @@ mod tests { // The terminal sorted run must not combine operations whose expiration // timestamps differ. With no active snapshot, retention keeps only the // newest canonical value. - let compacted = &db_state.tree.compacted.first().unwrap().sst_views; + let compacted = db_state.tree.compacted.first().unwrap().sst_views(); assert_eq!(compacted.len(), 1); let handle = compacted.first().unwrap(); @@ -3472,13 +3474,13 @@ mod tests { flush_type: FlushType::MemTable, }; - // write merge operations with the SAME ExpireAt timestamp at different clock times + // write merge operations with the SAME ExpireAtMillis timestamp at different clock times system_clock.set(100); db.merge_with_options( b"key1", b"a", &MergeOptions { - ttl: Ttl::ExpireAt(1000), + ttl: Ttl::ExpireAtMillis(1000), }, &WriteOptions { await_durable: false, @@ -3494,7 +3496,7 @@ mod tests { b"key1", b"b", &MergeOptions { - ttl: Ttl::ExpireAt(1000), + ttl: Ttl::ExpireAtMillis(1000), }, &WriteOptions { await_durable: false, @@ -3519,7 +3521,7 @@ mod tests { "compaction should have occurred" ); - let compacted = &db_state.tree.compacted.first().unwrap().sst_views; + let compacted = db_state.tree.compacted.first().unwrap().sst_views(); assert_eq!(compacted.len(), 1); let handle = compacted.first().unwrap(); @@ -3595,7 +3597,7 @@ mod tests { &[1; 16], value, &PutOptions { - ttl: Ttl::ExpireAt(10), + ttl: Ttl::ExpireAtMillis(10), }, &WriteOptions { await_durable: false, @@ -3611,7 +3613,7 @@ mod tests { &[2; 16], value, &PutOptions { - ttl: Ttl::ExpireAt(i64::MAX), + ttl: Ttl::ExpireAtMillis(i64::MAX), }, &WriteOptions { await_durable: false, @@ -3650,7 +3652,7 @@ mod tests { let db_state = db_state.expect("db was not compacted"); assert!(db_state.tree.last_compacted_l0_sst_view_id.is_some()); assert_eq!(db_state.tree.compacted.len(), 1); - let compacted = &db_state.tree.compacted.first().unwrap().sst_views; + let compacted = db_state.tree.compacted.first().unwrap().sst_views(); assert_eq!(compacted.len(), 1); let handle = compacted.first().unwrap(); let mut iter = SstIterator::new_borrowed_initialized( @@ -3688,7 +3690,7 @@ mod tests { } .into(); let mut options = db_options(Some(compactor_options())); - options.default_ttl = Some(50); + options.default_ttl_millis = Some(50); options .compactor_options .as_mut() @@ -3711,7 +3713,7 @@ mod tests { &[1; 16], value, &PutOptions { - ttl: Ttl::ExpireAfter(10), + ttl: Ttl::ExpireAfterMillis(10), }, &WriteOptions { await_durable: false, @@ -3758,7 +3760,7 @@ mod tests { &[1; 16], value, &PutOptions { - ttl: Ttl::ExpireAfter(80), + ttl: Ttl::ExpireAfterMillis(80), }, &WriteOptions { await_durable: false, @@ -3780,7 +3782,7 @@ mod tests { assert!(db_state.tree.last_compacted_l0_sst_view_id.is_some()); assert_eq!(db_state.tree.compacted.len(), 1); assert_eq!(db_state.last_l0_clock_tick, 70); - let compacted = &db_state.tree.compacted.first().unwrap().sst_views; + let compacted = db_state.tree.compacted.first().unwrap().sst_views(); assert_eq!(compacted.len(), 1); let handle = compacted.first().unwrap(); let mut iter = SstIterator::new_borrowed_initialized( @@ -3888,10 +3890,7 @@ mod tests { ..SsTableInfo::default() }, )); - let segment_sr = SortedRun { - id: 7, - sst_views: vec![segment_sr_view.clone()], - }; + let segment_sr = SortedRun::new(7, [segment_sr_view.clone()]); let mut core = ManifestCore::new(); core.segments = vec![Segment { @@ -3917,7 +3916,23 @@ mod tests { let expected = segment_l0.estimate_size() + segment_sr.estimate_size(); let actual = CompactorEventHandler::calculate_estimated_source_bytes(&compaction, &core); - assert_eq!(actual, expected); + assert_eq!(actual, Some(expected)); + } + + #[test] + fn test_calculate_estimated_source_bytes_returns_none_for_missing_source() { + let missing_l0 = Ulid::new(); + let compaction = Compaction::new( + Ulid::new(), + CompactionSpec::new(vec![SourceId::SstView(missing_l0)], 1), + ); + + let actual = CompactorEventHandler::calculate_estimated_source_bytes( + &compaction, + &ManifestCore::new(), + ); + + assert_eq!(actual, None); } #[tokio::test] @@ -4172,22 +4187,22 @@ mod tests { Arc::make_mut(&mut dirty.value.core.tree).l0 = VecDeque::from(vec![l0_view_newest, l0_view_oldest]); Arc::make_mut(&mut dirty.value.core.tree).compacted = vec![ - SortedRun { - id: 2, - sst_views: vec![SsTableView::identity(SsTableHandle::new( + SortedRun::new( + 2, + [SsTableView::identity(SsTableHandle::new( SsTableId::Compacted(Ulid::new()), SST_FORMAT_VERSION_LATEST, sr_info.clone(), ))], - }, - SortedRun { - id: 1, - sst_views: vec![SsTableView::identity(SsTableHandle::new( + ), + SortedRun::new( + 1, + [SsTableView::identity(SsTableHandle::new( SsTableId::Compacted(Ulid::new()), SST_FORMAT_VERSION_LATEST, sr_info.clone(), ))], - }, + ), ]; stored_manifest.update(dirty).await.unwrap(); @@ -4275,22 +4290,22 @@ mod tests { )); Arc::make_mut(&mut core.tree).l0 = VecDeque::from(vec![l0_view_first, l0_view_second]); Arc::make_mut(&mut core.tree).compacted = vec![ - SortedRun { - id: 5, - sst_views: vec![SsTableView::identity(SsTableHandle::new( + SortedRun::new( + 5, + [SsTableView::identity(SsTableHandle::new( SsTableId::Compacted(Ulid::from_parts(10, 0)), SST_FORMAT_VERSION_LATEST, sr_info.clone(), ))], - }, - SortedRun { - id: 2, - sst_views: vec![SsTableView::identity(SsTableHandle::new( + ), + SortedRun::new( + 2, + [SsTableView::identity(SsTableHandle::new( SsTableId::Compacted(Ulid::from_parts(11, 0)), SST_FORMAT_VERSION_LATEST, sr_info, ))], - }, + ), ]; let state = CompactorStateView { compactions: None, @@ -4372,22 +4387,22 @@ mod tests { last_compacted_l0_sst_id: None, l0: VecDeque::new(), compacted: vec![ - SortedRun { - id: 7, - sst_views: vec![SsTableView::identity(SsTableHandle::new( + SortedRun::new( + 7, + [SsTableView::identity(SsTableHandle::new( SsTableId::Compacted(Ulid::from_parts(70, 0)), SST_FORMAT_VERSION_LATEST, sr_info.clone(), ))], - }, - SortedRun { - id: 3, - sst_views: vec![SsTableView::identity(SsTableHandle::new( + ), + SortedRun::new( + 3, + [SsTableView::identity(SsTableHandle::new( SsTableId::Compacted(Ulid::from_parts(30, 0)), SST_FORMAT_VERSION_LATEST, sr_info, ))], - }, + ), ], }), }]; @@ -4427,14 +4442,14 @@ mod tests { first_entry: Some(Bytes::from_static(b"r")), ..SsTableInfo::default() }; - Arc::make_mut(&mut core.tree).compacted = vec![SortedRun { - id: 9, - sst_views: vec![SsTableView::identity(SsTableHandle::new( + Arc::make_mut(&mut core.tree).compacted = vec![SortedRun::new( + 9, + [SsTableView::identity(SsTableHandle::new( SsTableId::Compacted(Ulid::from_parts(90, 0)), SST_FORMAT_VERSION_LATEST, sr_info, ))], - }]; + )]; let state = CompactorStateView { compactions: None, manifest: VersionedManifest::from_manifest(0, Manifest::initial(core)), @@ -4463,14 +4478,14 @@ mod tests { first_entry: Some(Bytes::from_static(b"a")), ..SsTableInfo::default() }; - Arc::make_mut(&mut core.tree).compacted = vec![SortedRun { - id: 4, - sst_views: vec![SsTableView::identity(SsTableHandle::new( + Arc::make_mut(&mut core.tree).compacted = vec![SortedRun::new( + 4, + [SsTableView::identity(SsTableHandle::new( SsTableId::Compacted(Ulid::from_parts(40, 0)), SST_FORMAT_VERSION_LATEST, sr_info, ))], - }]; + )]; let state = CompactorStateView { compactions: None, manifest: VersionedManifest::from_manifest(0, Manifest::initial(core)), @@ -4502,22 +4517,22 @@ mod tests { ..SsTableInfo::default() }; Arc::make_mut(&mut core.tree).compacted = vec![ - SortedRun { - id: 8, - sst_views: vec![SsTableView::identity(SsTableHandle::new( + SortedRun::new( + 8, + [SsTableView::identity(SsTableHandle::new( SsTableId::Compacted(Ulid::from_parts(80, 0)), SST_FORMAT_VERSION_LATEST, info.clone(), ))], - }, - SortedRun { - id: 4, - sst_views: vec![SsTableView::identity(SsTableHandle::new( + ), + SortedRun::new( + 4, + [SsTableView::identity(SsTableHandle::new( SsTableId::Compacted(Ulid::from_parts(40, 0)), SST_FORMAT_VERSION_LATEST, info.clone(), ))], - }, + ), ]; core.segment_extractor_name = Some("test".into()); core.segments = vec![ @@ -4527,14 +4542,14 @@ mod tests { last_compacted_l0_sst_view_id: None, last_compacted_l0_sst_id: None, l0: VecDeque::new(), - compacted: vec![SortedRun { - id: 3, - sst_views: vec![SsTableView::identity(SsTableHandle::new( + compacted: vec![SortedRun::new( + 3, + [SsTableView::identity(SsTableHandle::new( SsTableId::Compacted(Ulid::from_parts(30, 0)), SST_FORMAT_VERSION_LATEST, info.clone(), ))], - }], + )], }), }, Segment { @@ -4544,22 +4559,22 @@ mod tests { last_compacted_l0_sst_id: None, l0: VecDeque::new(), compacted: vec![ - SortedRun { - id: 9, - sst_views: vec![SsTableView::identity(SsTableHandle::new( + SortedRun::new( + 9, + [SsTableView::identity(SsTableHandle::new( SsTableId::Compacted(Ulid::from_parts(90, 0)), SST_FORMAT_VERSION_LATEST, info.clone(), ))], - }, - SortedRun { - id: 6, - sst_views: vec![SsTableView::identity(SsTableHandle::new( + ), + SortedRun::new( + 6, + [SsTableView::identity(SsTableHandle::new( SsTableId::Compacted(Ulid::from_parts(60, 0)), SST_FORMAT_VERSION_LATEST, info, ))], - }, + ), ], }), }, @@ -4603,14 +4618,14 @@ mod tests { first_entry: Some(Bytes::from_static(b"x")), ..SsTableInfo::default() }; - Arc::make_mut(&mut core.tree).compacted = vec![SortedRun { - id: 5, - sst_views: vec![SsTableView::identity(SsTableHandle::new( + Arc::make_mut(&mut core.tree).compacted = vec![SortedRun::new( + 5, + [SsTableView::identity(SsTableHandle::new( SsTableId::Compacted(Ulid::from_parts(50, 0)), SST_FORMAT_VERSION_LATEST, info.clone(), ))], - }]; + )]; core.segment_extractor_name = Some("test".into()); core.segments = vec![ // L0-only: still skipped because L0 SSTs are ineligible inputs. @@ -4926,7 +4941,9 @@ mod tests { let completed = compaction .clone() .with_status(CompactionStatus::Compacted) - .with_output_ssts(result.sst_views.iter().map(|v| v.sst.clone()).collect()) + .with_output_ssts( + result.sst_views().iter().map(|v| v.sst.clone()).collect(), + ) .with_ctx(None); dirty.value.insert(completed); match stored.update(dirty).await { @@ -4995,7 +5012,7 @@ mod tests { .compacted .first() .unwrap() - .sst_views + .sst_views() .iter() .map(|view| view.sst.id.unwrap_compacted_id()) .collect(); @@ -5170,10 +5187,8 @@ mod tests { .value .core; Arc::make_mut(&mut core.tree).l0 = VecDeque::from([l0.clone()]); - Arc::make_mut(&mut core.tree).compacted = vec![SortedRun { - id: 1, - sst_views: vec![sr_first.clone(), sr_last.clone()], - }]; + Arc::make_mut(&mut core.tree).compacted = + vec![SortedRun::new(1, [sr_first.clone(), sr_last.clone()])]; let compaction_id = Ulid::new(); fixture @@ -5207,7 +5222,7 @@ mod tests { assert_eq!(output.id, 2); assert_eq!( output - .sst_views + .sst_views() .iter() .map(|view| view.id) .collect::>(), @@ -5242,10 +5257,7 @@ mod tests { .value .core; Arc::make_mut(&mut core.tree).l0 = VecDeque::from([l0.clone()]); - Arc::make_mut(&mut core.tree).compacted = vec![SortedRun { - id: 1, - sst_views: vec![sr_first, sr_last], - }]; + Arc::make_mut(&mut core.tree).compacted = vec![SortedRun::new(1, [sr_first, sr_last])]; let compaction_id = Ulid::new(); fixture .handler @@ -5442,7 +5454,7 @@ mod tests { // Build the handler and trigger a ticker to pick up the pre-existing Submitted entry. let scheduler = Arc::new(MockScheduler::new()); let rand = Arc::new(DbRand::default()); - let recorder = slatedb_common::metrics::MetricsRecorderHelper::noop(); + let recorder = MetricsRecorderHelper::noop(); let compactor_stats = Arc::new(CompactionStats::new(&recorder)); let mut handler = CompactorEventHandler::new( manifest_store, @@ -5734,10 +5746,7 @@ mod tests { // Root tree holds SR(7) — the global max. The segment-targeted spec // below proposes dst=3, which is above the segment's local max (0) // but below the global max. - Arc::make_mut(&mut core.tree).compacted = vec![SortedRun { - id: 7, - sst_views: Vec::new(), - }]; + Arc::make_mut(&mut core.tree).compacted = vec![SortedRun::new(7, [])]; core.segments = vec![Segment { prefix: prefix.clone(), tree: Arc::new(LsmTreeState { @@ -5779,10 +5788,7 @@ mod tests { .manifest_mut_for_test() .value .core; - Arc::make_mut(&mut core.tree).compacted = vec![SortedRun { - id: 7, - sst_views: Vec::new(), - }]; + Arc::make_mut(&mut core.tree).compacted = vec![SortedRun::new(7, [])]; core.segments = vec![Segment { prefix: prefix.clone(), tree: Arc::new(LsmTreeState { @@ -5832,10 +5838,7 @@ mod tests { last_compacted_l0_sst_view_id: None, last_compacted_l0_sst_id: None, l0: VecDeque::from(vec![make_view(l0_view)]), - compacted: vec![SortedRun { - id: 7, - sst_views: Vec::new(), - }], + compacted: vec![SortedRun::new(7, [])], }), }]; @@ -6019,10 +6022,7 @@ mod tests { .manifest_mut_for_test() .value .core; - Arc::make_mut(&mut core.tree).compacted = vec![SortedRun { - id: 99, - sst_views: Vec::new(), - }]; + Arc::make_mut(&mut core.tree).compacted = vec![SortedRun::new(99, [])]; let prefix = Bytes::from_static(b"seg/"); core.segments = vec![Segment { prefix: prefix.clone(), @@ -6062,10 +6062,7 @@ mod tests { .core; // Place SR(7) in the root tree. The segment-targeted spec below uses 7 as // its destination but does not list it among its sources. - Arc::make_mut(&mut core.tree).compacted = vec![SortedRun { - id: 7, - sst_views: Vec::new(), - }]; + Arc::make_mut(&mut core.tree).compacted = vec![SortedRun::new(7, [])]; // Seed SR(99) into the segment so the source-existence check passes and // destination-overwrite is the rejection reason. let prefix = Bytes::from_static(b"seg/"); @@ -6075,10 +6072,7 @@ mod tests { last_compacted_l0_sst_view_id: None, last_compacted_l0_sst_id: None, l0: VecDeque::new(), - compacted: vec![SortedRun { - id: 99, - sst_views: Vec::new(), - }], + compacted: vec![SortedRun::new(99, [])], }), }]; @@ -6275,7 +6269,7 @@ mod tests { .tree .compacted .first() - .is_some_and(|sr| sr.sst_views.len() == 1) + .is_some_and(|sr| sr.sst_views().len() == 1) } /// If a clock is provided, it will be advanced the clock by 60 seconds on each iteration to @@ -6456,7 +6450,10 @@ mod tests { sr.is_some(), "output SR {destination} not found in manifest" ); - assert_eq!(sr.unwrap().sst_views.first().unwrap().sst.id, output_sst.id); + assert_eq!( + sr.unwrap().sst_views().first().unwrap().sst.id, + output_sst.id + ); // given: a Compacted SR0→SR1 compaction to validate the SR source path removed when not in L0 let sr1_output_sst = fake_output_sst(); @@ -6488,7 +6485,7 @@ mod tests { let sr1 = core2.tree.compacted.iter().find(|sr| sr.id == 1); assert!(sr1.is_some(), "SR 1 should exist"); assert_eq!( - sr1.unwrap().sst_views.first().unwrap().sst.id, + sr1.unwrap().sst_views().first().unwrap().sst.id, sr1_output_sst.id ); let stored2 = fixture @@ -6629,11 +6626,9 @@ mod tests { assert!(c.worker().is_none(), "worker should be cleared"); // and: the reclamation is counted. - let reclaimed = slatedb_common::metrics::lookup_metric( - &fixture.test_recorder, - crate::compactor::stats::JOBS_RECLAIMED, - ) - .expect("metric not found"); + let reclaimed = + slatedb_common::metrics::lookup_metric(&fixture.test_recorder, stats::JOBS_RECLAIMED) + .expect("metric not found"); assert_eq!(reclaimed, 1, "one job should be counted as reclaimed"); } @@ -6645,11 +6640,8 @@ mod tests { let mut fixture = CompactorEventHandlerTestFixture::new().await; let claimed_count = || { - slatedb_common::metrics::lookup_metric( - &fixture.test_recorder, - crate::compactor::stats::JOBS_CLAIMED, - ) - .expect("metric not found") + slatedb_common::metrics::lookup_metric(&fixture.test_recorder, stats::JOBS_CLAIMED) + .expect("metric not found") }; // given: a job already claimed (Running) before the coordinator's first tick. diff --git a/slatedb/src/compactor_executor.rs b/slatedb/src/compactor_executor.rs index cadd28bc9e..ac58dcc44f 100644 --- a/slatedb/src/compactor_executor.rs +++ b/slatedb/src/compactor_executor.rs @@ -338,9 +338,7 @@ impl TokioCompactionExecutorInner { }; let sst_iter_options = SstIteratorOptions { max_fetch_tasks: self.options.max_fetch_tasks, - blocks_to_fetch: self - .table_store - .bytes_to_blocks(self.options.bytes_to_fetch), + target_bytes_to_fetch: self.options.bytes_to_fetch, cache_blocks: false, // don't clobber the cache cache_metadata: false, eager_spawn: true, @@ -770,16 +768,13 @@ impl TokioCompactionExecutorInner { ); return Err(SlateDBError::CompactorExecutorFailed); } - Ok(SortedRun { - id: destination, - sst_views: output_ssts - .into_iter() - .map(|sst| { - let id = self.rand.rng().gen_ulid(self.clock.as_ref()); - SsTableView::new(id, sst.clone()) - }) - .collect(), - }) + Ok(SortedRun::new( + destination, + output_ssts.into_iter().map(|sst| { + let id = self.rand.rng().gen_ulid(self.clock.as_ref()); + SsTableView::new(id, sst.clone()) + }), + )) } /// Runs the merge for one key range of a compaction job and returns the @@ -866,6 +861,9 @@ impl TokioCompactionExecutorInner { let total_bytes = start_bytes_processed + all_iter.bytes_processed(); progress(total_bytes, &output_ssts); } + + // Keep cached compaction work cooperative. + tokio::task::coop::consume_budget().await; } // Drain the in-flight close, then flush the final partial SST. Order @@ -1572,10 +1570,10 @@ mod tests { sr_ssts.extend(ssts); all_entries.extend(entries.iter().cloned()); } - sorted_runs.push(SortedRun { - id: sr_id as u32, - sst_views: sr_ssts.into_iter().map(SsTableView::identity).collect(), - }); + sorted_runs.push(SortedRun::new( + sr_id as u32, + sr_ssts.into_iter().map(SsTableView::identity), + )); } } @@ -1799,7 +1797,7 @@ mod tests { .unwrap(); let mut expected_entries = Vec::new(); - for view in &full_run.sst_views { + for view in full_run.sst_views() { let mut iter = SstIterator::new( SstView::Owned( Box::new(SsTableView::identity(view.sst.clone())), @@ -1851,7 +1849,7 @@ mod tests { .unwrap(); let mut resumed_entries = Vec::new(); - for view in &resumed_run.sst_views { + for view in resumed_run.sst_views() { let mut iter = SstIterator::new( SstView::Owned( Box::new(SsTableView::identity(view.sst.clone())), @@ -1878,7 +1876,7 @@ mod tests { /// runs can be compared for byte-identical merged output. async fn read_run_entries(table_store: &Arc, run: &SortedRun) -> Vec { let mut entries = Vec::new(); - for sst in &run.sst_views { + for sst in run.sst_views() { let mut iter = SstIterator::new( SstView::Borrowed(sst, BytesRange::from(..)), table_store.clone(), @@ -2129,7 +2127,7 @@ mod tests { .sum(); assert_eq!( final_output, - split.sst_views.len(), + split.sst_views().len(), "final snapshot should capture every output SST" ); } @@ -2198,7 +2196,7 @@ mod tests { ); for sst in snapshot.iter().flat_map(|s| s.output_ssts()) { assert!( - resumed.sst_views.iter().any(|v| v.sst.id == sst.id), + resumed.sst_views().iter().any(|v| v.sst.id == sst.id), "previously recorded subcompaction output SST was not reused" ); } @@ -2209,7 +2207,7 @@ mod tests { if index == snapshots.len() - 1 { let recorded: usize = snapshot.iter().map(|s| s.output_ssts().len()).sum(); assert_eq!( - resumed.sst_views.len(), + resumed.sst_views().len(), recorded, "resuming a completed compaction must not produce new SSTs" ); @@ -2790,7 +2788,7 @@ mod tests { // then: multiple output SSTs were produced (proving real boundaries and // background closes ran) ... - let result_ssts = &result.sst_views; + let result_ssts = result.sst_views(); assert!( result_ssts.len() >= 2, "expected multiple output SSTs, got {}", @@ -2798,7 +2796,7 @@ mod tests { ); // ... and the merged output preserves every key in ascending order. let mut read_back = Vec::new(); - for view in result_ssts { + for view in result_ssts.iter() { let mut iter = SstIterator::new( SstView::Owned( Box::new(SsTableView::identity(view.sst.clone())), @@ -2977,8 +2975,8 @@ mod tests { .await .unwrap(); - assert_eq!(1, result.sst_views.len()); - let sst = result.sst_views[0].clone(); + assert_eq!(1, result.sst_views().len()); + let sst = result.sst_views()[0].clone(); let mut iter = SstIterator::new( SstView::Borrowed(&sst, BytesRange::from(..)), table_store.clone(), @@ -3174,8 +3172,8 @@ mod tests { let result = ctx.run_compaction(vec![l0], true, None).await.unwrap(); // Verify the output SST - assert_eq!(1, result.sst_views.len()); - let sst = result.sst_views[0].clone(); + assert_eq!(1, result.sst_views().len()); + let sst = result.sst_views()[0].clone(); let mut iter = SstIterator::new( SstView::Borrowed(&sst, BytesRange::from(..)), table_store.clone(), diff --git a/slatedb/src/compactor_state.rs b/slatedb/src/compactor_state.rs index ad51aa7b65..ebbc186c64 100644 --- a/slatedb/src/compactor_state.rs +++ b/slatedb/src/compactor_state.rs @@ -526,8 +526,8 @@ impl Compaction { let mut sst_views = self.get_l0_sst_views(db_state); sst_views.extend( self.get_sorted_runs(db_state) - .into_iter() - .flat_map(|sr| sr.sst_views), + .iter() + .flat_map(|sr| sr.sst_views().iter().cloned()), ); sst_views.sort_by(|left, right| { left.compacted_effective_range() @@ -542,10 +542,7 @@ impl Compaction { .intersect(pair[1].compacted_effective_range()) .is_none() })) - .then_some(SortedRun { - id: destination, - sst_views, - }) + .then_some(SortedRun::new(destination, sst_views)) } /// The stable id (ULID) used to track this compaction across messages and attempts. @@ -681,9 +678,9 @@ impl CompactionsCore { self } - /// Returns an iterator over all recent compactions. Recent compactions include all - /// active (submitted or running) compactions as well as the most recently finished - /// compaction (failed or completed). + /// Returns all compactions retained in this state. Persisted state contains all active + /// compactions and the most recently finished one. Process-local state may also retain + /// older terminal entries until its next successful write. pub(crate) fn recent_compactions(&self) -> impl Iterator { self.recent_compactions.values() } @@ -804,8 +801,8 @@ impl Compactions { let latest_finished = self .core .recent_compactions - .iter() - .filter_map(|(_, c)| { + .values() + .filter_map(|c| { if c.status().finished() { Some(c.id()) } else { @@ -824,7 +821,8 @@ impl Compactions { /// /// This is the in-memory view that a single compactor task uses to: /// - keep a fresh `DirtyManifest` (view of `CoreDbState`), -/// - track in-flight compactions by id (ULID). +/// - track in-flight compactions by id (ULID), and +/// - retain terminal tombstones until their `.compactions` write succeeds. pub struct CompactorState { manifest: DirtyObject, compactions: DirtyObject, @@ -921,11 +919,11 @@ impl CompactorState { // For compactions not in local state (Vacant), accept new submissions, // worker-completed results, and retained terminal entries. On a write // conflict, the retry path reloads and merges the persisted `.compactions` - // object before writing again. At that point, local state may already have - // committed or pruned an entry while persisted state still contains an older - // Compacted/Completed/Failed view. Insert those entries, let the commit path - // resolve stale Compacted entries via `validate_compaction`, and let the compactions - // write path resolve stale terminal entries via `retain_active_and_last_finished`. + // object before writing again. Process-local terminal entries are deliberately + // retained until that write succeeds, so an older remote state for the same id + // is treated as a stale transition rather than resurrected as active work. + // Insert genuinely absent entries and let the commit path resolve stale + // Compacted entries via `validate_compaction`. // // Scheduled/Running are different: they carry no finished output and require prior // coordinator ownership, so seeing them absent from local state remains anomalous. @@ -963,11 +961,10 @@ impl CompactorState { } } - let mut merged_compactions = Compactions { + let merged_compactions = Compactions { compactor_epoch: self.compactions.value.compactor_epoch, core: CompactionsCore::new().with_compactions(merged), }; - merged_compactions.retain_active_and_last_finished(); remote_compactions.value = merged_compactions; self.set_compactions(remote_compactions); } @@ -1053,7 +1050,13 @@ impl CompactorState { Ok(()) } - /// Mutates a compaction in place if it exists, then trims retained state. + /// Mutates a compaction in place if it exists. + /// + /// Terminal entries remain in process-local state until the next successful + /// `.compactions` write. They act as tombstones during conflict retries so + /// an older persisted `Submitted` or `Compacted` record cannot be resurrected. + /// The persisted value is still trimmed by + /// [`crate::compactor_state_protocols::CompactorStateWriter`]. pub(crate) fn update_compaction(&mut self, compaction_id: &Ulid, f: F) where F: FnOnce(&mut Compaction), @@ -1067,7 +1070,6 @@ impl CompactorState { { f(compaction); } - self.compactions.value.retain_active_and_last_finished(); } /// Applies the effects of a finished compaction to the in-memory manifest. @@ -1310,10 +1312,8 @@ mod tests { let sr_last = bounded_sst_view(3, b"z", b"z"); let mut db_state = ManifestCore::new(); Arc::make_mut(&mut db_state.tree).l0 = VecDeque::from([l0.clone()]); - Arc::make_mut(&mut db_state.tree).compacted = vec![SortedRun { - id: 1, - sst_views: vec![sr_first.clone(), sr_last.clone()], - }]; + Arc::make_mut(&mut db_state.tree).compacted = + vec![SortedRun::new(1, [sr_first.clone(), sr_last.clone()])]; let compaction = Compaction::new( Ulid::new(), CompactionSpec::new(vec![SstView(l0.id), SourceId::SortedRun(1)], 2), @@ -1326,7 +1326,7 @@ mod tests { assert_eq!(output.id, 2); assert_eq!( output - .sst_views + .sst_views() .iter() .map(|view| view.id) .collect::>(), @@ -1340,10 +1340,7 @@ mod tests { let sr_view = bounded_sst_view(2, b"m", b"z"); let mut db_state = ManifestCore::new(); Arc::make_mut(&mut db_state.tree).l0 = VecDeque::from([l0.clone()]); - Arc::make_mut(&mut db_state.tree).compacted = vec![SortedRun { - id: 1, - sst_views: vec![sr_view], - }]; + Arc::make_mut(&mut db_state.tree).compacted = vec![SortedRun::new(1, [sr_view])]; let compaction = Compaction::new( Ulid::new(), CompactionSpec::new(vec![SstView(l0.id), SourceId::SortedRun(1)], 2), @@ -1664,11 +1661,7 @@ mod tests { .expect("failed to add compaction"); // when: - let compacted_ssts = before_compaction.tree.l0.iter().cloned().collect(); - let sr = SortedRun { - id: 0, - sst_views: compacted_ssts, - }; + let sr = SortedRun::new(0, before_compaction.tree.l0.iter().cloned()); state.finish_compaction(compaction_id, sr.clone()); // then: @@ -1679,14 +1672,14 @@ mod tests { assert_eq!(state.db_state().tree.l0.len(), 0); assert_eq!(state.db_state().tree.compacted.len(), 1); assert_eq!(state.db_state().tree.compacted.first().unwrap().id, sr.id); - let expected_ids: Vec = sr.sst_views.iter().map(|h| h.sst.id).collect(); + let expected_ids: Vec = sr.sst_views().iter().map(|h| h.sst.id).collect(); let found_ids: Vec = state .db_state() .tree .compacted .first() .unwrap() - .sst_views + .sst_views() .iter() .map(|h| h.sst.id) .collect(); @@ -1728,10 +1721,7 @@ mod tests { .add_compaction(Compaction::new(compaction_id, spec)) .expect("failed to add compaction"); - let sr = SortedRun { - id: 0, - sst_views: before_compaction.tree.l0.iter().cloned().collect(), - }; + let sr = SortedRun::new(0, before_compaction.tree.l0.iter().cloned()); state.finish_compaction(compaction_id, sr); let external_dbs = &state.manifest().value.external_dbs; @@ -1762,11 +1752,7 @@ mod tests { .expect("failed to add compaction"); // when: - let compacted_ssts = before_compaction.tree.l0.iter().cloned().collect(); - let sr = SortedRun { - id: 0, - sst_views: compacted_ssts, - }; + let sr = SortedRun::new(0, before_compaction.tree.l0.iter().cloned()); state.finish_compaction(compaction_id, sr); // then: @@ -1851,10 +1837,7 @@ mod tests { .expect("failed to add compaction"); state.finish_compaction( compaction_id, - SortedRun { - id: 0, - sst_views: vec![original_l0s.back().unwrap().clone()], - }, + SortedRun::new(0, [original_l0s.back().unwrap().clone()]), ); // open a new db and write another l0 let db = build_db(os.clone(), rt.handle()); @@ -1918,10 +1901,7 @@ mod tests { .expect("failed to add compaction"); state.finish_compaction( compaction_id, - SortedRun { - id: 0, - sst_views: original_l0s.clone().into(), - }, + SortedRun::new(0, original_l0s.iter().cloned()), ); assert_eq!(state.db_state().tree.l0.len(), 0); // open a new db and write another l0 @@ -2140,7 +2120,7 @@ mod tests { let original_l0s = &state.db_state().clone().tree.l0; let original_srs = &state.db_state().clone().tree.compacted; // L0: from 4th onward (index > 2) - let l0_sources = original_l0s.iter().skip(3).map(|h| SourceId::SstView(h.id)); + let l0_sources = original_l0s.iter().skip(3).map(|h| SstView(h.id)); // SRs: first 3 (index < 3) let sr_sources = original_srs @@ -2212,7 +2192,7 @@ mod tests { fn sorted_run_to_description(sr: &SortedRun) -> SortedRunDescription { SortedRunDescription { id: sr.id, - ssts: sr.sst_views.iter().map(|h| h.sst.id).collect(), + ssts: sr.sst_views().iter().map(|h| h.sst.id).collect(), } } @@ -2232,7 +2212,7 @@ mod tests { } fn build_l0_compaction(ssts: &VecDeque, dst: u32) -> CompactionSpec { - let sources = ssts.iter().map(|h| SourceId::SstView(h.id)).collect(); + let sources = ssts.iter().map(|h| SstView(h.id)).collect(); CompactionSpec::new(sources, dst) } @@ -2250,14 +2230,8 @@ mod tests { // Add a named segment with two compacted SRs (ids 5 and 3, list-position // ordered newest-first as per ManifestCore conventions). let prefix = Bytes::from_static(b"hour=12/"); - let sr5 = SortedRun { - id: 5, - sst_views: Vec::new(), - }; - let sr3 = SortedRun { - id: 3, - sst_views: Vec::new(), - }; + let sr5 = SortedRun::new(5, []); + let sr3 = SortedRun::new(3, []); let segment = Segment { prefix: prefix.clone(), tree: Arc::new(LsmTreeState { @@ -2283,10 +2257,7 @@ mod tests { .expect("failed to add compaction"); // Finish the compaction with a fresh output SR. - let output = SortedRun { - id: 7, - sst_views: Vec::new(), - }; + let output = SortedRun::new(7, []); state.finish_compaction(compaction_id, output); // The segment's compacted list now holds only the new SR(7). @@ -2317,10 +2288,7 @@ mod tests { last_compacted_l0_sst_view_id: None, last_compacted_l0_sst_id: None, l0: VecDeque::new(), - compacted: vec![SortedRun { - id: 7, - sst_views: Vec::new(), - }], + compacted: vec![SortedRun::new(7, [])], }), }]; @@ -2333,10 +2301,7 @@ mod tests { // Segment dropped after submission, before finish. state.manifest.value.core.segments = Vec::new(); - let output = SortedRun { - id: 7, - sst_views: Vec::new(), - }; + let output = SortedRun::new(7, []); state.finish_compaction(compaction_id, output); let compaction = state @@ -2358,10 +2323,7 @@ mod tests { // Seed real sources so the source-isolation check passes for both // submissions. Root tree gets SR(99); segment "seg/" gets SR(100). - Arc::make_mut(&mut state.manifest.value.core.tree).compacted = vec![SortedRun { - id: 99, - sst_views: Vec::new(), - }]; + Arc::make_mut(&mut state.manifest.value.core.tree).compacted = vec![SortedRun::new(99, [])]; let prefix = Bytes::from_static(b"seg/"); state.manifest.value.core.segments = vec![Segment { prefix: prefix.clone(), @@ -2369,10 +2331,7 @@ mod tests { last_compacted_l0_sst_view_id: None, last_compacted_l0_sst_id: None, l0: VecDeque::new(), - compacted: vec![SortedRun { - id: 100, - sst_views: Vec::new(), - }], + compacted: vec![SortedRun::new(100, [])], }), }]; @@ -2401,10 +2360,7 @@ mod tests { // Seed SR(7) in the root tree so the source-isolation check passes // for the first submission; the spec rewrites that SR (destination=7). - Arc::make_mut(&mut state.manifest.value.core.tree).compacted = vec![SortedRun { - id: 7, - sst_views: Vec::new(), - }]; + Arc::make_mut(&mut state.manifest.value.core.tree).compacted = vec![SortedRun::new(7, [])]; let first_id = rand.rng().gen_ulid(system_clock.as_ref()); let first = CompactionSpec::new(vec![SourceId::SortedRun(7)], 7); @@ -2442,13 +2398,13 @@ mod tests { }]; let first_id = rand.rng().gen_ulid(system_clock.as_ref()); - let first = CompactionSpec::drain_segment(prefix.clone(), vec![SourceId::SstView(l0_a.id)]); + let first = CompactionSpec::drain_segment(prefix.clone(), vec![SstView(l0_a.id)]); state .add_compaction(Compaction::new(first_id, first)) .expect("first drain must register"); let second_id = rand.rng().gen_ulid(system_clock.as_ref()); - let second = CompactionSpec::drain_segment(prefix, vec![SourceId::SstView(l0_b.id)]); + let second = CompactionSpec::drain_segment(prefix, vec![SstView(l0_b.id)]); let err = state .add_compaction(Compaction::new(second_id, second)) .expect_err("second drain on same segment must be rejected"); @@ -2488,13 +2444,13 @@ mod tests { ]; let first_id = rand.rng().gen_ulid(system_clock.as_ref()); - let first = CompactionSpec::drain_segment(prefix_a, vec![SourceId::SstView(l0_a.id)]); + let first = CompactionSpec::drain_segment(prefix_a, vec![SstView(l0_a.id)]); state .add_compaction(Compaction::new(first_id, first)) .expect("first drain must register"); let second_id = rand.rng().gen_ulid(system_clock.as_ref()); - let second = CompactionSpec::drain_segment(prefix_b, vec![SourceId::SstView(l0_b.id)]); + let second = CompactionSpec::drain_segment(prefix_b, vec![SstView(l0_b.id)]); state .add_compaction(Compaction::new(second_id, second)) .expect("drain on a different segment must register"); @@ -2520,7 +2476,7 @@ mod tests { }]; let compaction_id = rand.rng().gen_ulid(system_clock.as_ref()); - let spec = CompactionSpec::drain_segment(prefix, vec![SourceId::SstView(l0.id)]); + let spec = CompactionSpec::drain_segment(prefix, vec![SstView(l0.id)]); state .add_compaction(Compaction::new(compaction_id, spec)) .expect("drain submission must register"); @@ -2559,10 +2515,7 @@ mod tests { // Two L0s (newest first) and one SR in the segment. let l0_newer = drain_test_view(2); let l0_older = drain_test_view(1); - let sr = SortedRun { - id: 5, - sst_views: Vec::new(), - }; + let sr = SortedRun::new(5, []); let prefix = Bytes::from_static(b"hour=10/"); state.manifest.value.core.segments = vec![Segment { prefix: prefix.clone(), @@ -2578,8 +2531,8 @@ mod tests { let spec = CompactionSpec::drain_segment( prefix.clone(), vec![ - SourceId::SstView(l0_newer.id), - SourceId::SstView(l0_older.id), + SstView(l0_newer.id), + SstView(l0_older.id), SourceId::SortedRun(sr.id), ], ); @@ -2626,8 +2579,7 @@ mod tests { }]; let compaction_id = rand.rng().gen_ulid(system_clock.as_ref()); - let spec = - CompactionSpec::drain_segment(prefix.clone(), vec![SourceId::SstView(l0_observed.id)]); + let spec = CompactionSpec::drain_segment(prefix.clone(), vec![SstView(l0_observed.id)]); state .add_compaction(Compaction::new(compaction_id, spec)) .expect("drain compaction must register"); diff --git a/slatedb/src/compactor_state_protocols.rs b/slatedb/src/compactor_state_protocols.rs index 142b58fce9..8005158127 100644 --- a/slatedb/src/compactor_state_protocols.rs +++ b/slatedb/src/compactor_state_protocols.rs @@ -290,6 +290,11 @@ impl CompactorStateWriter { /// Persists the current compactions state to the compactions store and refreshes the /// local dirty object with the latest version. /// + /// Process-local state retains every terminal transition until this write succeeds. + /// Only the outgoing value is trimmed to the latest terminal entry. This preserves + /// terminal entries as tombstones across conflict retries while keeping the persisted + /// `.compactions` object bounded. + /// /// ## Returns /// - `Ok(())` when compactions are successfully written. /// - `SlateDBError` if an unrecoverable error occurs. @@ -307,8 +312,10 @@ impl CompactorStateWriter { } Err(err) if err.is_sequenced_write_conflict() => { // Merge the latest remote state (e.g. a worker's Compacted write) into - // the coordinator's view before retrying. Without this, retrying with a stale - // desired_value could silently overwrite worker progress. + // the coordinator's untrimmed local view before retrying. Without this, + // retrying with a stale desired_value could silently overwrite worker + // progress. Local terminal entries also prevent stale remote active states + // for the same ids from being resurrected. self.load_compactions().await?; desired_value = self.state.compactions().value.clone(); desired_value.retain_active_and_last_finished(); @@ -890,6 +897,82 @@ mod tests { assert_eq!(final_id, start_id + 2); } + #[tokio::test] + async fn test_write_compactions_safely_does_not_resurrect_terminal_on_conflict() { + let object_store: Arc = Arc::new(InMemory::new()); + let manifest_store = Arc::new(ManifestStore::new( + &Path::from(ROOT), + Arc::clone(&object_store), + )); + let compactions_store = Arc::new(CompactionsStore::new( + &Path::from(ROOT), + Arc::clone(&object_store), + )); + let system_clock: Arc = Arc::new(DefaultSystemClock::new()); + + StoredManifest::create_new_db( + manifest_store.clone(), + ManifestCore::new(), + system_clock.clone(), + ) + .await + .unwrap(); + + let mut writer = CompactorStateWriter::new( + manifest_store, + compactions_store.clone(), + system_clock, + &CompactorOptions::default(), + Arc::new(DbRand::new(7)), + ) + .await + .unwrap(); + + let completed_id = Ulid::from_parts(10, 0); + let failed_id = Ulid::from_parts(20, 0); + let spec = CompactionSpec::new(vec![], 0); + writer + .state + .insert_compaction_for_test(Compaction::new(completed_id, spec.clone())); + writer + .state + .insert_compaction_for_test(Compaction::new(failed_id, spec.clone())); + writer.state.update_compaction(&completed_id, |compaction| { + compaction.set_status(CompactionStatus::Completed) + }); + writer.state.update_compaction(&failed_id, |compaction| { + compaction.set_status(CompactionStatus::Failed) + }); + + // Advance the remote version with the pre-transition state of the completed + // compaction. This forces the coordinator's first write to conflict and reload. + let mut external = StoredCompactions::load(compactions_store.clone()) + .await + .unwrap(); + let mut dirty = external.prepare_dirty().unwrap(); + dirty.value.insert(Compaction::new(completed_id, spec)); + external.update(dirty).await.unwrap(); + + writer.write_compactions_safely().await.unwrap(); + + let persisted = compactions_store.read_latest_compactions().await.unwrap(); + assert!( + persisted + .recent_compactions() + .all(|compaction| compaction.id() != completed_id), + "stale Submitted compaction was resurrected" + ); + assert_eq!( + persisted + .recent_compactions() + .find(|compaction| compaction.id() == failed_id) + .expect("latest terminal compaction was not retained") + .status(), + CompactionStatus::Failed + ); + assert_eq!(writer.state.active_compactions().count(), 0); + } + #[tokio::test] async fn test_write_compactions_safely_retries_on_boundary_conflict() { let object_store: Arc = Arc::new(InMemory::new()); diff --git a/slatedb/src/config.rs b/slatedb/src/config.rs index 2ed475f12d..f49eafb0af 100644 --- a/slatedb/src/config.rs +++ b/slatedb/src/config.rs @@ -281,6 +281,20 @@ pub enum DurabilityLevel { Memory, } +/// Options for tracing a read operation. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct TracingOptions { + pub trace_id: String, +} + +impl TracingOptions { + pub fn new(trace_id: impl Into) -> Self { + Self { + trace_id: trace_id.into(), + } + } +} + /// Configuration for client read operations. `ReadOptions` is supplied for each /// read call and controls the behavior of the read. #[derive(Clone, Debug)] @@ -298,6 +312,8 @@ pub struct ReadOptions { /// Optional context forwarded to custom filter policies; ignored by /// built-in filters. See [`FilterContext`]. pub filter_context: Option, + /// Optional caller-provided tracing settings. + pub tracing_options: Option, } impl Default for ReadOptions { @@ -307,6 +323,7 @@ impl Default for ReadOptions { dirty: false, cache_blocks: true, filter_context: None, + tracing_options: None, } } } @@ -340,6 +357,13 @@ impl ReadOptions { ..self } } + + pub fn with_tracing_options(self, tracing_options: Option) -> Self { + Self { + tracing_options, + ..self + } + } } #[derive(Clone, Debug)] pub struct ScanOptions { @@ -350,9 +374,10 @@ pub struct ScanOptions { /// Whether to include dirty data in the scan. "dirty" means that the data is not considered /// as "committed" yet, whose seq number is greater than the last committed seq number. pub dirty: bool, - /// The number of bytes to read ahead. The value is rounded up to the nearest - /// block size when fetching from object storage. The default is 1, which - /// rounds up to one block. + /// The target number of bytes to fetch in a single request while iterating over SSTs. + /// Each fetch will read the minimum number of blocks such that the resulting read is at least + /// this size or reaches the end of the file. The default is 1, which results in each fetch + /// reading one block. pub read_ahead_bytes: usize, /// Whether or not fetched data blocks should be cached. SST indexes, /// filters, and stats are cached independently of this setting. @@ -368,6 +393,8 @@ pub struct ScanOptions { /// Only consulted for `scan_prefix` today. Plain range scans do not /// evaluate SST filters, so this field has no effect on `scan`. pub filter_context: Option, + /// Optional caller-provided tracing settings. + pub tracing_options: Option, } impl Default for ScanOptions { @@ -381,6 +408,7 @@ impl Default for ScanOptions { max_fetch_tasks: 1, order: IterationOrder::Ascending, filter_context: None, + tracing_options: None, } } } @@ -432,10 +460,17 @@ impl ScanOptions { ..self } } + + pub fn with_tracing_options(self, tracing_options: Option) -> Self { + Self { + tracing_options, + ..self + } + } } /// Enum representing the type of flush to perform. -#[derive(Clone)] +#[derive(Clone, Debug)] pub enum FlushType { /// Freeze the active memtable [crate::mem_table::KVTable] and write /// all immutable memtable entries (including the formerly active @@ -461,12 +496,44 @@ impl Default for FlushOptions { } } +/// Options controlling how a database is closed. +#[derive(Clone, Debug)] +pub struct CloseOptions { + /// The type of flush to perform before closing. + /// + /// When `None`, no final flush is triggered. Memtables already being + /// flushed continue through the existing shutdown pipeline, and writes + /// that are not durable may be lost. When set to `Some` flushes the + /// database in accordance with the specified [`FlushType`] + /// Defaults to `Some(FlushType::MemTable)`. + pub flush_type: Option, +} + +impl Default for CloseOptions { + fn default() -> Self { + Self { + flush_type: Some(FlushType::MemTable), + } + } +} + +impl CloseOptions { + /// Configure the type of flush to perform before closing. + pub fn with_flush_type(mut self, flush_type: Option) -> Self { + self.flush_type = flush_type; + self + } +} + /// Configuration for client write operations. `WriteOptions` is supplied for each /// write call and controls the behavior of the write. #[derive(Clone, Debug)] pub struct WriteOptions { - /// Whether `put` calls should block until the write has been durably committed - /// to the DB. + /// Whether the write call waits for the returned handle to become durable before returning. + /// + /// Defaults to `true` for compatibility with the fork's v0.15 write contract. Callers that + /// want v0.16's non-blocking write behavior can set this to `false` and await the returned + /// [`crate::WriteHandle`] when durability is required. pub await_durable: bool, #[cfg(dst)] /// Force the current timestamp for DST operations. See #719 for details. @@ -479,7 +546,6 @@ pub struct WriteOptions { } impl Default for WriteOptions { - /// Create a new `WriteOptions`` with `await_durable` set to `true`. fn default() -> Self { Self { await_durable: true, @@ -503,25 +569,29 @@ pub struct PutOptions { } impl PutOptions { - pub(crate) fn expire_ts_from(&self, default: Option, now: i64) -> Option { + pub(crate) fn expire_ts_from( + &self, + default_ttl_millis: Option, + now_millis: i64, + ) -> Option { match self.ttl { - Ttl::Default => match default { + Ttl::Default => match default_ttl_millis { None => None, - Some(default_ttl) => Self::checked_expire_ts(now, default_ttl), + Some(default_ttl_millis) => Self::checked_expire_ts(now_millis, default_ttl_millis), }, Ttl::NoExpiry => None, - Ttl::ExpireAfter(ttl) => Self::checked_expire_ts(now, ttl), - Ttl::ExpireAt(ts) => Some(ts), + Ttl::ExpireAfterMillis(ttl_millis) => Self::checked_expire_ts(now_millis, ttl_millis), + Ttl::ExpireAtMillis(timestamp_millis) => Some(timestamp_millis), } } - fn checked_expire_ts(now: i64, ttl: u64) -> Option { + fn checked_expire_ts(now_millis: i64, ttl_millis: u64) -> Option { // for overflow, we will just assume no TTL - if ttl > i64::MAX as u64 { + if ttl_millis > i64::MAX as u64 { return None; }; - let expire_ts = now + (ttl as i64); - if expire_ts < now { + let expire_ts = now_millis + (ttl_millis as i64); + if expire_ts < now_millis { return None; }; @@ -541,25 +611,29 @@ pub struct MergeOptions { impl MergeOptions { // TODO(agavra): deduplicate this with PutOptions::expire_ts_from - pub(crate) fn expire_ts_from(&self, default: Option, now: i64) -> Option { + pub(crate) fn expire_ts_from( + &self, + default_ttl_millis: Option, + now_millis: i64, + ) -> Option { match self.ttl { - Ttl::Default => match default { + Ttl::Default => match default_ttl_millis { None => None, - Some(default_ttl) => Self::checked_expire_ts(now, default_ttl), + Some(default_ttl_millis) => Self::checked_expire_ts(now_millis, default_ttl_millis), }, Ttl::NoExpiry => None, - Ttl::ExpireAfter(ttl) => Self::checked_expire_ts(now, ttl), - Ttl::ExpireAt(ts) => Some(ts), + Ttl::ExpireAfterMillis(ttl_millis) => Self::checked_expire_ts(now_millis, ttl_millis), + Ttl::ExpireAtMillis(timestamp_millis) => Some(timestamp_millis), } } - fn checked_expire_ts(now: i64, ttl: u64) -> Option { + fn checked_expire_ts(now_millis: i64, ttl_millis: u64) -> Option { // for overflow, we will just assume no TTL - if ttl > i64::MAX as u64 { + if ttl_millis > i64::MAX as u64 { return None; }; - let expire_ts = now + (ttl as i64); - if expire_ts < now { + let expire_ts = now_millis + (ttl_millis as i64); + if expire_ts < now_millis { return None; }; @@ -567,14 +641,25 @@ impl MergeOptions { } } +/// Time-to-live policy applied to an inserted value or merge operand. +/// +/// TTL durations are expressed in milliseconds. Absolute expiration timestamps are +/// expressed as milliseconds since the Unix epoch. +/// +/// Expiration is applied during compaction and is therefore best effort; an expired +/// value may remain visible until compaction processes it. #[non_exhaustive] #[derive(Clone, Default, PartialEq, Debug)] pub enum Ttl { + /// Use [`Settings::default_ttl_millis`]. #[default] Default, + /// Store the value without an expiration. NoExpiry, - ExpireAfter(u64), - ExpireAt(i64), + /// Expire the value after the specified number of milliseconds. + ExpireAfterMillis(u64), + /// Expire the value at the specified Unix timestamp in milliseconds. + ExpireAtMillis(i64), } /// Defines the scope targeted by a given checkpoint. If set to All, then the checkpoint will @@ -780,11 +865,12 @@ pub struct Settings { #[serde(default)] pub metric_level: MetricLevel, - /// The default time-to-live (TTL) for insertions (note that re-inserting a key - /// with any value will update the TTL to use the default_ttl) + /// The default time-to-live (TTL), in milliseconds, for insertions (note that + /// re-inserting a key with any value will update the TTL to use + /// `default_ttl_millis`). /// /// Default: no TTL (insertions will remain until deleted) - pub default_ttl: Option, + pub default_ttl_millis: Option, /// Maximum logical-discriminator payload retained by one transaction. #[serde(default = "default_max_transaction_conflict_metadata_bytes")] @@ -844,7 +930,7 @@ impl std::fmt::Debug for Settings { ) .field("garbage_collector_options", &self.garbage_collector_options) .field("metric_level", &self.metric_level) - .field("default_ttl", &self.default_ttl) + .field("default_ttl_millis", &self.default_ttl_millis) .field( "max_transaction_conflict_metadata_bytes", &self.max_transaction_conflict_metadata_bytes, @@ -1065,7 +1151,7 @@ impl Settings { } impl Provider for Settings { - fn metadata(&self) -> figment::Metadata { + fn metadata(&self) -> Metadata { Metadata::named("SlateDb configuration options") } @@ -1097,7 +1183,7 @@ impl Default for Settings { object_store_cache_options: ObjectStoreCacheOptions::default(), garbage_collector_options: Some(GarbageCollectorOptions::default()), metric_level: MetricLevel::default(), - default_ttl: None, + default_ttl_millis: None, max_transaction_conflict_metadata_bytes: DEFAULT_MAX_TRANSACTION_CONFLICT_METADATA_BYTES, max_retained_conflict_metadata_bytes: DEFAULT_MAX_RETAINED_CONFLICT_METADATA_BYTES, @@ -1146,6 +1232,7 @@ impl WalReplaySettings { "wal_replay.max_inflight_bytes must fit in a 32-bit semaphore permit count".into(), )); } + self.encoded_byte_limit()?; Ok(()) } @@ -1230,13 +1317,12 @@ pub struct DbReaderOptions { /// local filesystem, mirroring the behaviour of `Db`. pub object_store_cache_options: ObjectStoreCacheOptions, - /// When true, skip WAL replay entirely. The reader will only see data that has been - /// compacted into L0 or lower levels. This is useful for read-heavy workloads that - /// don't need to see the most recent uncommitted writes and want to minimize the + /// When true, skip WAL replay entirely, in every reader mode. The reader reads no WAL + /// when it opens or when it refreshes its state, so it only sees data that has been + /// flushed to L0 or lower levels. This is useful for read-heavy workloads that + /// don't need to see the most recent writes and want to minimize the /// cost of opening many readers. /// - /// WAL replay is also skipped in [`crate::DbReaderMode::Checkpoint`] mode. - /// /// When combined with a reader mode that polls manifests, the reader will still see newly /// compacted data as manifests are updated. /// @@ -1444,9 +1530,10 @@ pub struct CompactionWorkerOptions { /// compaction. Higher values can improve throughput but use more resources. pub max_fetch_tasks: usize, - /// Number of bytes to fetch in a single read-ahead request while iterating - /// over input SSTs during compaction. The value is rounded up to the nearest - /// block size when fetching from object storage. The default is 2MiB. + /// The target number of bytes to fetch in a single request while iterating over + /// input SSTs during compaction. Each fetch will read the minimum number of blocks + /// such that the resulting read is at least this size or reaches the end of the file. + /// The default is 2MiB. /// /// This pairs with [`CompactionWorkerOptions::max_fetch_tasks`]: /// `bytes_to_fetch` is the size of each read-ahead request while @@ -1702,7 +1789,7 @@ impl GarbageCollectorOptions { /// Default options for the garbage collector for a directory. /// -/// By default, the garbage collector will run every minute and deletes files +/// By default, the garbage collector will run every 10 minutes and deletes files /// that are at least 5 minutes old. impl Default for GarbageCollectorDirectoryOptions { fn default() -> Self { @@ -2108,6 +2195,12 @@ object_store_cache_options: assert_eq!(options.max_fetch_tasks, 1); } + #[test] + fn test_default_tracing_options_are_none() { + assert!(ReadOptions::default().tracing_options.is_none()); + assert!(ScanOptions::default().tracing_options.is_none()); + } + #[test] fn test_scan_options_with_max_fetch_tasks() { let options = ScanOptions::default().with_max_fetch_tasks(4); @@ -2120,6 +2213,32 @@ object_store_cache_options: assert!(!options.cache_blocks); } + #[test] + fn test_read_options_with_tracing_options() { + let options = + ReadOptions::default().with_tracing_options(Some(TracingOptions::new("read-trace"))); + assert_eq!( + options + .tracing_options + .as_ref() + .map(|tracing_options| tracing_options.trace_id.as_str()), + Some("read-trace") + ); + } + + #[test] + fn test_scan_options_with_tracing_options() { + let options = + ScanOptions::default().with_tracing_options(Some(TracingOptions::new("scan-trace"))); + assert_eq!( + options + .tracing_options + .as_ref() + .map(|tracing_options| tracing_options.trace_id.as_str()), + Some("scan-trace") + ); + } + #[test] fn test_size_tiered_compaction_scheduler_options_roundtrip() { let options = SizeTieredCompactionSchedulerOptions { @@ -2140,7 +2259,7 @@ object_store_cache_options: fn should_return_exact_timestamp_for_put_expire_at() { // given let opts = PutOptions { - ttl: Ttl::ExpireAt(12345), + ttl: Ttl::ExpireAtMillis(12345), }; // when @@ -2154,7 +2273,7 @@ object_store_cache_options: fn should_ignore_default_ttl_for_put_expire_at() { // given let opts = PutOptions { - ttl: Ttl::ExpireAt(12345), + ttl: Ttl::ExpireAtMillis(12345), }; // when @@ -2168,7 +2287,7 @@ object_store_cache_options: fn should_allow_past_timestamp_for_put_expire_at() { // given let opts = PutOptions { - ttl: Ttl::ExpireAt(50), + ttl: Ttl::ExpireAtMillis(50), }; // when @@ -2182,7 +2301,7 @@ object_store_cache_options: fn should_return_exact_timestamp_for_merge_expire_at() { // given let opts = MergeOptions { - ttl: Ttl::ExpireAt(12345), + ttl: Ttl::ExpireAtMillis(12345), }; // when @@ -2194,9 +2313,9 @@ object_store_cache_options: #[test] fn should_return_deterministic_expire_ts_for_expire_at() { - // given: same ExpireAt value used at different times + // given: same ExpireAtMillis value used at different times let opts = PutOptions { - ttl: Ttl::ExpireAt(99999), + ttl: Ttl::ExpireAtMillis(99999), }; // when diff --git a/slatedb/src/db.rs b/slatedb/src/db.rs index 883b18e2b4..d0e2d58bb2 100644 --- a/slatedb/src/db.rs +++ b/slatedb/src/db.rs @@ -23,8 +23,8 @@ pub use crate::db_status::{DbStatus, SegmentPrefix}; use crate::db_cache::CacheTarget; -use crate::db_cache_manager; -use std::ops::Range; +use std::fmt; +use std::future::Future; use std::sync::Arc; use bytes::Bytes; @@ -37,7 +37,7 @@ use crate::db_transaction::DbTransaction; use crate::dispatcher::MessageHandlerExecutor; use crate::garbage_collector::GC_TASK_NAME; use crate::transaction_manager::IsolationLevel; -use crate::CloseReason; +use crate::{db_cache_manager, CloseReason}; use log::{debug, info, trace, warn}; use parking_lot::RwLock; use std::time::Duration; @@ -48,8 +48,8 @@ use crate::bytes_range::{ByteRangeBounds, BytesRange}; use crate::cached_object_store::CachedObjectStore; use crate::clock::MonotonicClock; use crate::config::{ - FlushOptions, FlushType, MergeOptions, PutOptions, ReadOptions, ScanOptions, Settings, - WriteOptions, + CloseOptions, FlushOptions, FlushType, MergeOptions, PutOptions, ReadOptions, ScanOptions, + Settings, WriteOptions, }; use crate::db_common::extract_segment_prefix; use crate::db_iter::{DbIterator, DbRecencyIterator}; @@ -70,15 +70,15 @@ use crate::tablestore::TableStore; use crate::transaction_manager::TransactionManager; use crate::types::KeyValue; use crate::utils::{format_bytes_si, SafeSender, WatchableOnceCellReader}; -use crate::wal_replay::{ExactWalReplayIterator, ExactWalReplaySource, WalReplayOptions}; +use crate::wal_replay::{WalReplayIterator, WalReplayOptions}; use crate::{DbCacheManagerOps, DbMetadataOps, DbReadOps, DbWriteOps}; use slatedb_common::clock::SystemClock; use slatedb_common::metrics::MetricsRecorderHelper; use slatedb_common::DbRand; use slatedb_txn_obj::DirtyObject; -use crate::db_status::{ClosedResultWriter, DbStatusManager}; -use crate::wal::{WalEvent, WalObserver, WalStatus}; +use crate::db_status::{ClosedResultWriter, DbStatusManager, DurabilityWaiter}; +use crate::wal::{WalEvent, WalIterator, WalObserver, WalStatus}; pub use builder::DbBuilder; pub use builder::DbReaderBuilder; @@ -133,7 +133,7 @@ impl DbInner { wal_observer: Box, recorder: MetricsRecorderHelper, fp_registry: Arc, - merge_operator: Option, + merge_operator: Option, status_manager: Arc, segment_extractor: Option>, ) -> Result { @@ -331,15 +331,12 @@ impl DbInner { self.maybe_apply_backpressure().await?; self.write_notifier.send(batch_msg)?; - // TODO: this can be modified as awaiting the last_durable_seq watermark & fatal error. - let write_handle = rx.await??; - if options.await_durable { let seq = write_handle.seq; let mut status_subscription = self.status_manager.subscribe(); let status = status_subscription - .wait_for(|s| s.durable_seq >= seq || s.close_reason.is_some()) + .wait_for(|status| status.durable_seq >= seq || status.close_reason.is_some()) .await .map_err(|_| SlateDBError::Closed)?; if status.durable_seq < seq { @@ -351,7 +348,6 @@ impl DbInner { return Err(SlateDBError::InvalidDBState); } } - Ok(write_handle) } @@ -507,11 +503,8 @@ impl DbInner { } } - async fn replay_wal(&self, wal_id_range: Range) -> Result<(), SlateDBError> { + async fn replay_wal(&self, wal_iterator: Box) -> Result<(), SlateDBError> { let replay_started = self.system_clock.now(); - let replay_range_start = wal_id_range.start; - let replay_range_end = wal_id_range.end; - let replay_wal_count = replay_range_end.saturating_sub(replay_range_start); let mut replayed_entries = 0_u64; let mut replayed_bytes = 0_u64; let mut current_memtable_wal_id = self @@ -522,6 +515,7 @@ impl DbInner { .value .core .replay_after_wal_id; + let replay_range_start = current_memtable_wal_id.saturating_add(1); let writer_epoch = self.state.read().state().manifest.value.writer_epoch; fail_point!( Arc::clone(&self.fp_registry), @@ -531,21 +525,17 @@ impl DbInner { ); let replay_options = WalReplayOptions { - prefetch: self.settings.wal_replay, max_memtable_bytes: self.settings.l0_sst_size_bytes, - min_seq: None, - source: ExactWalReplaySource::WriterOpen, - task_scope: None, + ..Default::default() }; let db_state = self.state.read().state().core().clone(); - let mut replay_iter = ExactWalReplayIterator::range( - wal_id_range, + let mut replay_iter = WalReplayIterator::for_wal_iterator( + wal_iterator, &db_state, replay_options, Arc::clone(&self.table_store), - ) - .await?; + )?; loop { let replayed_table = match replay_iter.next().await { @@ -555,12 +545,12 @@ impl DbInner { // indicate that a newer writer has advanced `replay_after_wal_id` and // the GC has removed this WAL entry. Check the latest manifest's // writer_epoch to see if this client is fenced. - Err(err) if err.has_object_store_not_found() => { + Err(SlateDBError::WalTruncated(wal_id)) => { self.memtable_flusher.refresh_manifest().await?; if self.state.read().state().manifest.value.writer_epoch > writer_epoch { return Err(SlateDBError::Fenced); } - return Err(err); + return Err(SlateDBError::WalTruncated(wal_id)); } Err(err) => return Err(err), }; @@ -609,6 +599,8 @@ impl DbInner { let guard = self.state.read(); self.status_manager .report_memtable_segments(collect_touched_segments(&guard.view())); + let replay_range_end = current_memtable_wal_id.saturating_add(1); + let replay_wal_count = replay_range_end.saturating_sub(replay_range_start); info!( "SlateDB WAL replay completed [writer_epoch={}, replay_start_wal_id={}, replay_end_wal_id={}, replay_wal_count={}, last_replayed_wal_id={}, replayed_entries={}, replayed_bytes={}, elapsed_ms={}]", writer_epoch, @@ -788,31 +780,31 @@ impl Db { /// } /// ``` pub async fn close(&self) -> Result<(), crate::Error> { - let should_flush = match self.status().close_reason { + self.close_with_options(CloseOptions::default()).await + } + + /// Close the database with custom options. + /// + pub async fn close_with_options(&self, options: CloseOptions) -> Result<(), crate::Error> { + let flush_type = match self.status().close_reason { // If already closed, don't close again. Some(CloseReason::Clean) => return Err(SlateDBError::Closed.into()), // If in failed state, allow close, but don't flush since the database // might be in a bad state. Note that multiple close() calls will always // run when in a failed state (vs. a clean closure, which will return // Error::Closed(CloseReason::Clean) on subsequent calls). - Some(_) => false, - // Flush outstanding writes if the database is still open. - None => true, + Some(_) => None, + // Flush outstanding writes if the database is still open and the + // caller requested a final flush. + None => options.flush_type, }; // Mark the database as closed before flushing. self.inner.status_manager.write_result(Ok(())); - let result = if should_flush { - // Flush memtables to L0 so that the WAL does not need to be - // replayed on the next startup. + let result = if let Some(flush_type) = flush_type { self.inner - .flush( - FlushOptions { - flush_type: FlushType::MemTable, - }, - false, - ) + .flush(FlushOptions { flush_type }, false) .await .map_err(Into::into) .inspect_err(|e| warn!("failed to flush db during close [error={:?}]", e)) @@ -1454,7 +1446,11 @@ impl Db { .map_err(Into::into) } - /// Write a value into the database with default `WriteOptions`. + /// Write a value into the database with default `PutOptions` and + /// `WriteOptions`. + /// + /// The default [`WriteOptions`] wait for the write to become durable in + /// object storage before returning. /// /// ## Arguments /// - `key`: the key to write @@ -1490,6 +1486,11 @@ impl Db { /// Write a value into the database with custom `PutOptions` and `WriteOptions`. /// + /// Durability behavior is controlled by [`WriteOptions::await_durable`]. + /// When it is `false`, call [`WriteHandle::await_durable`] on the returned + /// handle to wait for this write, or [`Db::flush`] to flush all pending + /// writes. + /// /// ## Arguments /// - `key`: the key to write /// - `value`: the value to write @@ -1535,6 +1536,9 @@ impl Db { /// this form when the caller already holds the data as [`Bytes`] (e.g. /// from a prior read, a zero-copy buffer pool, or a client that produces /// [`Bytes`] directly). + /// + /// The default [`WriteOptions`] wait for the write to become durable in + /// object storage before returning. pub async fn put_bytes(&self, key: Bytes, value: Bytes) -> Result { self.put_bytes_with_options(key, value, &PutOptions::default(), &WriteOptions::default()) .await @@ -1543,6 +1547,11 @@ impl Db { /// Write a value into the database using owned [`Bytes`] with custom /// `PutOptions` and `WriteOptions`. See [`Db::put_bytes`] for why this /// form exists. + /// + /// Durability behavior is controlled by [`WriteOptions::await_durable`]. + /// When it is `false`, call [`WriteHandle::await_durable`] on the returned + /// handle to wait for this write, or [`Db::flush`] to flush all pending + /// writes. pub async fn put_bytes_with_options( &self, key: Bytes, @@ -1557,6 +1566,9 @@ impl Db { /// Delete a key from the database with default `WriteOptions`. /// + /// The default [`WriteOptions`] wait for the delete to become durable in + /// object storage before returning. + /// /// ## Arguments /// - `key`: the key to delete /// @@ -1586,6 +1598,11 @@ impl Db { /// Delete a key from the database with custom `WriteOptions`. /// + /// Durability behavior is controlled by [`WriteOptions::await_durable`]. + /// When it is `false`, call [`WriteHandle::await_durable`] on the returned + /// handle to wait for this delete, or [`Db::flush`] to flush all pending + /// writes. + /// /// ## Arguments /// - `key`: the key to delete /// - `options`: the write options to use @@ -1620,6 +1637,9 @@ impl Db { /// Merge a value into the database with default `MergeOptions` and `WriteOptions`. /// + /// The default [`WriteOptions`] wait for the merge to become durable in + /// object storage before returning. + /// /// Merge operations allow applications to bypass the traditional read/modify/write cycle /// by expressing partial updates using an associative operator. The merge operator must /// be configured when opening the database. @@ -1676,6 +1696,11 @@ impl Db { /// Merge a value into the database with custom `MergeOptions` and `WriteOptions`. /// + /// Durability behavior is controlled by [`WriteOptions::await_durable`]. + /// When it is `false`, call [`WriteHandle::await_durable`] on the returned + /// handle to wait for this merge, or [`Db::flush`] to flush all pending + /// writes. + /// /// Merge operations allow applications to bypass the traditional read/modify/write cycle /// by expressing partial updates using an associative operator. The merge operator must /// be configured when opening the database. @@ -1747,6 +1772,9 @@ impl Db { /// block other gets and writes until the batch is written to the WAL (or memtable if /// WAL is disabled). /// + /// The default [`WriteOptions`] wait for the batch to become durable in + /// object storage before returning. + /// /// ## Arguments /// - `batch`: the batch of put/delete operations to write /// @@ -1783,6 +1811,11 @@ impl Db { /// block other gets and writes until the batch is written to the WAL (or memtable if /// WAL is disabled). /// + /// Durability behavior is controlled by [`WriteOptions::await_durable`]. + /// When it is `false`, call [`WriteHandle::await_durable`] on the returned + /// handle to wait for this batch, or [`Db::flush`] to flush all pending + /// writes. + /// /// ## Arguments /// - `batch`: the batch of put/delete operations to write /// - `options`: the write options to use @@ -1822,8 +1855,8 @@ impl Db { .map_err(Into::into) } - /// Flush in-memory writes to disk. This function blocks until the in-memory - /// data has been durably written to object storage. + /// Flush in-memory writes to object storage. This function blocks until + /// the in-memory data has been durably written. /// /// If WAL is enabled, this method is equivalent to: /// `flush_with_options(FlushOptions { flush_type: FlushType::Wal })` @@ -1983,7 +2016,7 @@ impl Db { let env_vars = std::env::vars().map(|(key, value)| (key.to_ascii_lowercase(), value)); let (object_store, path) = parse_url_opts(&url, env_vars).map_err(SlateDBError::from)?; if !path.as_ref().is_empty() { - return Err(SlateDBError::InvalidObjectStorePath(path.to_string()))?; + return Err(SlateDBError::InvalidObjectStorePath(path.to_string()).into()); } Ok(Arc::from(object_store)) } @@ -2154,16 +2187,59 @@ impl DbCacheManagerOps for Db { } /// Handle returned from write operations, containing metadata about the write. +/// +/// Write operations return this handle after applying their configured +/// [`WriteOptions`]. When [`WriteOptions::await_durable`] is `false`, call +/// [`WriteHandle::await_durable`] to wait until the write is durable in object +/// storage. +/// /// This structure is designed to be extensible for future enhancements. -#[derive(Debug, Clone)] +#[derive(Clone)] pub struct WriteHandle { pub(crate) seq: u64, pub(crate) create_ts: i64, + durability_waiter: DurabilityWaiter, +} + +impl fmt::Debug for WriteHandle { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.debug_struct("WriteHandle") + .field("seq", &self.seq) + .field("create_ts", &self.create_ts) + .finish_non_exhaustive() + } } impl WriteHandle { - pub fn new(seq: u64, create_ts: i64) -> Self { - Self { seq, create_ts } + /// Creates a write handle that uses `durability_waiter` to determine when + /// the write is durable. + /// + /// This constructor is intended for custom [`DbWriteOps`] implementations + /// and test doubles whose writes are not immediately durable. + pub fn new(seq: u64, create_ts: i64, durability_waiter: F) -> Self + where + F: Fn() -> Fut + Send + Sync + 'static, + Fut: Future> + Send + 'static, + { + let durability_waiter: DurabilityWaiter = Arc::new(move |_| Box::pin(durability_waiter())); + + Self { + seq, + create_ts, + durability_waiter, + } + } + + pub(crate) fn new_with_waiter( + seq: u64, + create_ts: i64, + durability_waiter: DurabilityWaiter, + ) -> Self { + Self { + seq, + create_ts, + durability_waiter, + } } /// Returns the sequence number assigned to this write operation. @@ -2175,9 +2251,21 @@ impl WriteHandle { pub fn create_ts(&self) -> i64 { self.create_ts } + + /// Waits until this write has been durably persisted. + /// + /// If the database closes before the write becomes durable, this returns a + /// closed error carrying the database's [`CloseReason`]. + /// + /// # Errors + /// + /// Returns a closed error if the database closes first. + pub async fn await_durable(&self) -> Result<(), crate::Error> { + (self.durability_waiter)(self.seq).await + } } -/// Wraps [`WalObserver`] and injects a [`crate::wal_buffer::WalStatusListener`] +/// Wraps [`WalObserver`] and injects a [`crate::wal::WalStatusListener`] /// that updates the oracle and manifest, and drives cross-task notifications about wal events /// via a [`tokio::sync::watch`] channel. #[derive(Clone)] @@ -2262,7 +2350,7 @@ mod tests { use crate::config::DurabilityLevel::{Memory, Remote}; use crate::config::MetricLevel; use crate::config::{ - CheckpointOptions, CompactionWorkerOptions, CompactorOptions, + CheckpointOptions, CloseOptions, CompactionWorkerOptions, CompactorOptions, GarbageCollectorDirectoryOptions, GarbageCollectorOptions, ObjectStoreCacheOptions, PutOptions, ScanOptions, Settings, SstBlockSize, Ttl, WriteOptions, }; @@ -2290,8 +2378,7 @@ mod tests { OnDemandCompactionSchedulerSupplier, StringConcatMergeOperator, }; use crate::types::RowEntry; - use crate::wal::WalError; - use crate::wal_reader::WalReader; + use crate::wal::{SlateDbWalReaderBuilder, WalError, WalReader as _}; use crate::{proptest_util, test_utils, CloseReason, CompactorBuilder, KeyValue}; use async_trait::async_trait; use chrono::{TimeZone, Utc}; @@ -2303,7 +2390,8 @@ mod tests { use slatedb_common::clock::DefaultSystemClock; use slatedb_common::clock::MockSystemClock; use slatedb_common::metrics::{ - lookup_metric, lookup_metric_with_labels, DefaultMetricsRecorder, MetricValue, + lookup_metric, lookup_metric_with_labels, CounterFn, DefaultMetricsRecorder, GaugeFn, + HistogramFn, MetricValue, MetricsRecorder, NoopMetricsRecorder, UpDownCounterFn, }; use std::collections::BTreeMap; use std::collections::Bound::Included; @@ -2339,6 +2427,35 @@ mod tests { }) } + fn lookup_object_store_api_request_count( + recorder: &DefaultMetricsRecorder, + component: &'static str, + store_type: &'static str, + op: &'static str, + api: &'static str, + ) -> i64 { + lookup_metric_with_labels( + recorder, + OBJECT_STORE_REQUEST_COUNT, + &object_store_labels(component, store_type, op, api), + ) + .unwrap_or(0) + } + + fn lookup_object_store_api_histogram_count( + recorder: &DefaultMetricsRecorder, + component: &'static str, + store_type: &'static str, + op: &'static str, + api: &'static str, + ) -> u64 { + lookup_object_store_histogram_count( + recorder, + &object_store_labels(component, store_type, op, api), + ) + .unwrap_or(0) + } + fn lookup_object_store_op_request_count( recorder: &DefaultMetricsRecorder, component: &'static str, @@ -2359,12 +2476,7 @@ mod tests { apis.iter() .map(|api| { - lookup_metric_with_labels( - recorder, - OBJECT_STORE_REQUEST_COUNT, - &object_store_labels(component, store_type, op, api), - ) - .unwrap_or(0) + lookup_object_store_api_request_count(recorder, component, store_type, op, api) }) .sum() } @@ -2389,11 +2501,7 @@ mod tests { apis.iter() .map(|api| { - lookup_object_store_histogram_count( - recorder, - &object_store_labels(component, store_type, op, api), - ) - .unwrap_or(0) + lookup_object_store_api_histogram_count(recorder, component, store_type, op, api) }) .sum() } @@ -2775,7 +2883,7 @@ mod tests { kv_store.close().await.unwrap(); } - fn assert_value(entry: &crate::types::RowEntry, expected: &[u8]) { + fn assert_value(entry: &RowEntry, expected: &[u8]) { match &entry.value { crate::types::ValueDeletable::Value(v) => assert_eq!(v.as_ref(), expected), other => panic!("expected Value({expected:?}), got {other:?}"), @@ -3018,7 +3126,7 @@ mod tests { &PutOptions::default(), &WriteOptions { await_durable: false, - seqnum: 0, + ..Default::default() }, ) .await @@ -3211,6 +3319,7 @@ mod tests { dirty: false, cache_blocks: true, filter_context: None, + tracing_options: None, } ) .await @@ -3319,7 +3428,7 @@ mod tests { assert_eq!( lookup_metric( &metrics_recorder, - crate::wal_buffer::stats::WAL_BUFFER_FLUSHES + crate::wal_buffer_stats::WAL_BUFFER_FLUSHES ) .unwrap(), 0 @@ -3340,7 +3449,7 @@ mod tests { assert_eq!( lookup_metric( &metrics_recorder, - crate::wal_buffer::stats::WAL_BUFFER_FLUSHES + crate::wal_buffer_stats::WAL_BUFFER_FLUSHES ) .unwrap(), 1 @@ -3398,7 +3507,7 @@ mod tests { assert_eq!( lookup_metric( &metrics_recorder, - crate::wal_buffer::stats::WAL_BUFFER_FLUSHES + crate::wal_buffer_stats::WAL_BUFFER_FLUSHES ) .unwrap(), 0 @@ -3407,7 +3516,7 @@ mod tests { // Simulate a failed state (e.g. fenced). db.inner .status_manager - .write_result(Err(crate::error::SlateDBError::Fenced)); + .write_result(Err(SlateDBError::Fenced)); // close() should succeed but not flush when failed. db.close().await.unwrap(); @@ -3418,7 +3527,7 @@ mod tests { assert_eq!( lookup_metric( &metrics_recorder, - crate::wal_buffer::stats::WAL_BUFFER_FLUSHES + crate::wal_buffer_stats::WAL_BUFFER_FLUSHES ) .unwrap(), 0 @@ -3463,7 +3572,7 @@ mod tests { assert_eq!( lookup_metric( &metrics_recorder, - crate::wal_buffer::stats::WAL_BUFFER_FLUSHES + crate::wal_buffer_stats::WAL_BUFFER_FLUSHES ) .unwrap(), 0 @@ -3477,7 +3586,7 @@ mod tests { assert_eq!( lookup_metric( &metrics_recorder, - crate::wal_buffer::stats::WAL_BUFFER_FLUSHES + crate::wal_buffer_stats::WAL_BUFFER_FLUSHES ) .unwrap(), 1 @@ -3529,6 +3638,185 @@ mod tests { ); } + #[tokio::test] + async fn test_close_with_options_default_flushes_final_memtable() { + let object_store: Arc = Arc::new(InMemory::new()); + let mut settings = test_db_options(0, 1024, None); + settings.flush_interval = None; + let metrics_recorder = Arc::new(DefaultMetricsRecorder::new()); + let db = Db::builder( + "/tmp/test_close_with_options_default_flushes_final_memtable", + object_store, + ) + .with_settings(settings) + .with_metrics_recorder(metrics_recorder.clone()) + .build() + .await + .unwrap(); + + db.put_with_options( + b"test_key", + b"test_value", + &PutOptions::default(), + &WriteOptions { + await_durable: false, + ..Default::default() + }, + ) + .await + .unwrap(); + + assert_eq!( + lookup_metric(&metrics_recorder, crate::db_stats::L0_FLUSH_BYTES).unwrap_or(0), + 0 + ); + + db.close_with_options(CloseOptions::default()) + .await + .unwrap(); + + assert!( + lookup_metric(&metrics_recorder, crate::db_stats::L0_FLUSH_BYTES).unwrap() > 0, + "expected L0 flush during close with default options" + ); + } + + #[tokio::test] + async fn test_close_with_options_flushes_wal_only() { + let object_store: Arc = Arc::new(InMemory::new()); + let mut settings = test_db_options(0, 1024, None); + settings.flush_interval = None; + let path = "/tmp/test_close_with_options_flushes_wal_only"; + let db = Db::builder(path, object_store.clone()) + .with_settings(settings.clone()) + .build() + .await + .unwrap(); + + db.put_with_options( + b"test_key", + b"test_value", + &PutOptions::default(), + &WriteOptions { + await_durable: false, + ..Default::default() + }, + ) + .await + .unwrap(); + + db.close_with_options(CloseOptions::default().with_flush_type(Some(FlushType::Wal))) + .await + .unwrap(); + + let manifest = db.manifest(); + assert!( + manifest.manifest.core.replay_after_wal_id + 1 < manifest.manifest.core.next_wal_sst_id, + "expected a data WAL after the replay watermark" + ); + assert!( + manifest.manifest.core.tree.l0.is_empty(), + "expected no flushed memtables in the manifest" + ); + + let reopened = Db::builder(path, object_store) + .with_settings(settings) + .build() + .await + .unwrap(); + assert_eq!( + reopened.get(b"test_key").await.unwrap(), + Some(Bytes::from_static(b"test_value")) + ); + reopened.close().await.unwrap(); + } + + #[tokio::test] + async fn test_close_with_options_skips_final_flush() { + let object_store: Arc = Arc::new(InMemory::new()); + let mut settings = test_db_options(0, 1024, None); + settings.flush_interval = None; + let metrics_recorder = Arc::new(DefaultMetricsRecorder::new()); + let db = Db::builder( + "/tmp/test_close_with_options_skips_final_flush", + object_store, + ) + .with_settings(settings) + .with_metrics_recorder(metrics_recorder.clone()) + .build() + .await + .unwrap(); + + db.put_with_options( + b"test_key", + b"test_value", + &PutOptions::default(), + &WriteOptions { + await_durable: false, + ..Default::default() + }, + ) + .await + .unwrap(); + + db.close_with_options(CloseOptions::default().with_flush_type(None)) + .await + .unwrap(); + + assert_eq!( + lookup_metric(&metrics_recorder, crate::wal_buffer_stats::WAL_FLUSH_BYTES,) + .unwrap_or(0), + 0 + ); + assert_eq!( + lookup_metric(&metrics_recorder, crate::db_stats::L0_FLUSH_BYTES).unwrap_or(0), + 0 + ); + } + + #[cfg(feature = "wal_disable")] + #[tokio::test] + async fn test_close_without_flush_fails_pending_durability_wait() { + let object_store: Arc = Arc::new(InMemory::new()); + let mut settings = test_db_options(0, 1024, None); + settings.flush_interval = None; + settings.wal_enabled = false; + let db = Db::builder( + "/tmp/test_close_without_flush_fails_pending_durability_wait", + object_store, + ) + .with_settings(settings) + .build() + .await + .unwrap(); + + let handle = db + .put_with_options( + b"key", + b"value", + &PutOptions::default(), + &WriteOptions { + await_durable: false, + ..Default::default() + }, + ) + .await + .unwrap(); + + db.close_with_options(CloseOptions::default().with_flush_type(None)) + .await + .unwrap(); + + let error = tokio::time::timeout(Duration::from_secs(5), handle.await_durable()) + .await + .expect("durability wait remained blocked after close") + .expect_err("discarded write should not become durable"); + assert!(matches!( + error.kind(), + crate::ErrorKind::Closed(CloseReason::Clean) + )); + } + #[tokio::test] async fn test_memtable_write_bytes_matches_batch_payload() { let object_store: Arc = Arc::new(InMemory::new()); @@ -3649,7 +3937,7 @@ mod tests { .await .unwrap(); assert_eq!( - lookup_metric(&metrics_recorder, crate::wal_buffer::stats::WAL_FLUSH_BYTES) + lookup_metric(&metrics_recorder, crate::wal_buffer_stats::WAL_FLUSH_BYTES,) .unwrap_or(0), 0, ); @@ -3657,7 +3945,7 @@ mod tests { db.flush().await.unwrap(); let wal_bytes = - lookup_metric(&metrics_recorder, crate::wal_buffer::stats::WAL_FLUSH_BYTES).unwrap(); + lookup_metric(&metrics_recorder, crate::wal_buffer_stats::WAL_FLUSH_BYTES).unwrap(); let memtable_bytes = lookup_metric(&metrics_recorder, crate::db_stats::MEMTABLE_WRITE_BYTES).unwrap(); // WAL SST framing/footer makes the encoded payload at least as large as @@ -3683,7 +3971,7 @@ mod tests { tokio::time::timeout( Duration::from_secs(1), db.task_executor - .join_task(crate::wal_buffer::WAL_BUFFER_TASK_NAME), + .join_task(crate::wal::slatedb::writer::WAL_BUFFER_TASK_NAME), ) .await .expect("native WAL task should not run when the WAL is disabled") @@ -3846,7 +4134,7 @@ mod tests { async fn build_database_from_table( table: &BTreeMap, db_options: Settings, - await_durable: bool, + wait_for_durability: bool, ) -> Db { let object_store: Arc = Arc::new(InMemory::new()); let db = Db::builder("/tmp/test_kv_store", object_store) @@ -3857,7 +4145,7 @@ mod tests { test_utils::seed_database(&db, table, false).await.unwrap(); - if await_durable { + if wait_for_durability { db.flush().await.unwrap(); } @@ -4360,7 +4648,7 @@ mod tests { // the memtable will not be flushed to l0, and the test will hang // at this put_with_options call. let write_options = WriteOptions { - await_durable: true, + await_durable: false, ..Default::default() }; clock.set(10); @@ -4371,6 +4659,9 @@ mod tests { &write_options, ) .await + .unwrap() + .await_durable() + .await .unwrap(); let state = wait_for_manifest_condition( @@ -4436,10 +4727,22 @@ mod tests { for i in 0..3 { let key = [b'a' + i; 16]; let value = [b'b' + i; 50]; - kv_store.put(&key, &value).await.unwrap(); + kv_store + .put(&key, &value) + .await + .unwrap() + .await_durable() + .await + .unwrap(); let key = [b'j' + i; 16]; let value = [b'k' + i; 50]; - kv_store.put(&key, &value).await.unwrap(); + kv_store + .put(&key, &value) + .await + .unwrap() + .await_durable() + .await + .unwrap(); let db_state = wait_for_manifest_condition( &mut stored_manifest, |s| s.replay_after_wal_id > last_wal_id, @@ -4735,7 +5038,7 @@ mod tests { let mut settings = test_db_options(0, 1024, None); settings.flush_interval = None; - let kv_store = Db::builder(path, main_object_store) + let kv_store = Db::builder(path, Arc::clone(&main_object_store)) .with_settings(settings) .with_wal_object_store(wal_object_store.clone()) .build() @@ -4784,20 +5087,25 @@ mod tests { 0 ); - let wal_reader = WalReader::new(path, wal_object_store); - let wal_files = wal_reader.list(..).await.unwrap(); - assert_eq!(wal_files.len(), 2); // first file is the fencing operation + let wal_reader = SlateDbWalReaderBuilder::new() + .with_object_store(main_object_store) + .with_wal_object_store(wal_object_store) + .with_path(Path::from(path)) + .build() + .unwrap(); + let tail = wal_reader.last_wal_file_id(0).await.unwrap(); + assert_eq!(tail, 2); // first file is the fencing operation let mut rows = Vec::new(); - let mut wal_iter = wal_files[1] // second file contains the actual write - .iterator() + let mut wal_iter = wal_reader + .iterator((1..tail.checked_add(1).unwrap()).into()) .await .expect("expected successful WAL iterator call"); - while let Some(entry) = wal_iter + while let Some(batch) = wal_iter .next() .await .expect("expected successful WAL rows read") { - rows.push(entry); + rows.extend(batch.rows); } assert_eq!(rows.len(), 1); let row = &rows[0]; @@ -5825,6 +6133,9 @@ mod tests { kv_store .put("foo".as_bytes(), "bar".as_bytes()) .await + .unwrap() + .await_durable() + .await .unwrap(); fail_parallel::cfg(fp_registry.clone(), "write-wal-sst-io-error", "pause").unwrap(); kv_store @@ -5876,6 +6187,9 @@ mod tests { kv_store .put("foo".as_bytes(), "bar".as_bytes()) .await + .unwrap() + .await_durable() + .await .unwrap(); fail_parallel::cfg(fp_registry.clone(), "write-wal-sst-io-error", "pause").unwrap(); kv_store @@ -5934,6 +6248,9 @@ mod tests { kv_store .put("key3".as_bytes(), "committed3".as_bytes()) .await + .unwrap() + .await_durable() + .await .unwrap(); // Pause WAL writes to prevent new writes from being committed @@ -6042,11 +6359,21 @@ mod tests { // write a few keys that will result in memtable flushes let key1 = [b'a'; 32]; let value1 = [b'b'; 96]; - db.put(key1, value1).await.unwrap(); + db.put(key1, value1) + .await + .unwrap() + .await_durable() + .await + .unwrap(); next_wal_id += 1; let key2 = [b'c'; 32]; let value2 = [b'd'; 96]; - db.put(key2, value2).await.unwrap(); + db.put(key2, value2) + .await + .unwrap() + .await_durable() + .await + .unwrap(); next_wal_id += 1; let reader = Db::builder(path, object_store.clone()) @@ -6130,6 +6457,7 @@ mod tests { let value1 = [b'b'; 96]; let result = db.put(&key1, &value1).await; assert!(result.is_ok(), "Failed to write key1"); + result.unwrap().await_durable().await.unwrap(); assert_eq!( db.inner.wal_observer.status().unwrap().last_flushed_wal_id, 2 @@ -6199,7 +6527,21 @@ mod tests { ); fail_parallel::cfg(fp_registry.clone(), "write-wal-sst-io-error", "panic").unwrap(); - let result = db.put(b"foo", b"bar").await.unwrap_err(); + let result = db + .put_with_options( + b"foo", + b"bar", + &PutOptions::default(), + &WriteOptions { + await_durable: false, + ..Default::default() + }, + ) + .await + .unwrap() + .await_durable() + .await + .unwrap_err(); assert!(result.to_string().contains("background task panicked")); } @@ -6217,12 +6559,31 @@ mod tests { .unwrap(), ); // Trigger a WAL write and block until durable so WAL is written - db.put(b"foo", b"bar").await.unwrap(); + db.put(b"foo", b"bar") + .await + .unwrap() + .await_durable() + .await + .unwrap(); fail_parallel::cfg(fp_registry.clone(), "write-wal-sst-io-error", "panic").unwrap(); // Trigger a WAL write, which should not advance the manifest WAL ID - let result = db.put(b"foo", b"bar").await.unwrap_err(); + let result = db + .put_with_options( + b"foo", + b"bar", + &PutOptions::default(), + &WriteOptions { + await_durable: false, + ..Default::default() + }, + ) + .await + .unwrap() + .await_durable() + .await + .unwrap_err(); assert_eq!(result.kind(), crate::ErrorKind::Closed(CloseReason::Panic)); assert!(result .to_string() @@ -6292,15 +6653,67 @@ mod tests { .expect_err("close should error out due to WAL IO error"); } + #[tokio::test] + async fn test_constructed_write_handle_uses_durability_waiter() { + let waiter_called = Arc::new(AtomicBool::new(false)); + let waiter_called_clone = waiter_called.clone(); + let handle = WriteHandle::new(1, 0, move || { + let waiter_called = waiter_called_clone.clone(); + async move { + waiter_called.store(true, Ordering::SeqCst); + Ok(()) + } + }); + + handle.await_durable().await.unwrap(); + + assert!(waiter_called.load(Ordering::SeqCst)); + } + + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] + async fn test_write_handle_await_durable_waits_for_flush() { + let object_store: Arc = Arc::new(InMemory::new()); + let mut settings = test_db_options(0, 1024, None); + settings.flush_interval = None; + let db = Db::builder( + "/tmp/test_write_handle_await_durable_waits_for_flush", + object_store, + ) + .with_settings(settings) + .build() + .await + .unwrap(); + + let handle = db + .put_with_options( + b"foo", + b"bar", + &PutOptions::default(), + &WriteOptions { + await_durable: false, + ..Default::default() + }, + ) + .await + .unwrap(); + let durability_wait = tokio::spawn(async move { handle.await_durable().await }); + tokio::task::yield_now().await; + assert!(!durability_wait.is_finished()); + + db.flush().await.unwrap(); + durability_wait.await.unwrap().unwrap(); + db.close().await.unwrap(); + } + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] - async fn test_await_durable_write_returns_error_if_db_closes_before_durable() { + async fn test_write_handle_await_durable_returns_error_if_db_closes_first() { let fp_registry = Arc::new(FailPointRegistry::new()); let object_store: Arc = Arc::new(InMemory::new()); let mut settings = test_db_options(0, 1024, None); settings.flush_interval = None; let db = Arc::new( Db::builder( - "/tmp/test_await_durable_write_returns_error_if_db_closes_before_durable", + "/tmp/test_write_handle_await_durable_returns_error_if_db_closes_first", object_store, ) .with_settings(settings) @@ -6314,14 +6727,15 @@ mod tests { fail_parallel::cfg(fp_registry.clone(), "write-wal-sst-io-error", "pause").unwrap(); let write_db = db.clone(); let write_task = tokio::spawn(async move { - write_db + let handle = write_db .put_with_options( b"foo", b"bar", &PutOptions::default(), &WriteOptions::default(), ) - .await + .await?; + handle.await_durable().await }); tokio::time::timeout(Duration::from_secs(10), async { loop { @@ -6393,8 +6807,7 @@ mod tests { |s| { // compact after writing values. include in loop since the on demand scheduler // only runs once per `should_compact`, and memtables might still be getting - // flushed (await_durable in the put()'s above only wait for the writes to hit - // the WAL before returning). + // flushed (the put() calls above return before the writes become durable). should_compact_l0.store(true, Ordering::SeqCst); s.tree.last_compacted_l0_sst_view_id.is_some() && s.tree.l0.is_empty() }, @@ -6419,8 +6832,7 @@ mod tests { |s| { // compact after writing values. include in loop since the on demand scheduler // only runs once per `should_compact`, and memtables might still be getting - // flushed (await_durable in the put()'s above only wait for the writes to hit - // the WAL before returning). + // flushed (the put() calls above return before the writes become durable). should_compact_l0.store(true, Ordering::SeqCst); s.tree.last_compacted_l0_sst_view_id.is_some() && s.tree.l0.is_empty() }, @@ -6533,16 +6945,11 @@ mod tests { let path = "/tmp/test_kv_store"; async fn do_put(db: &Db, key: &[u8], val: &[u8]) -> Result { - db.put_with_options( - key, - val, - &PutOptions::default(), - &WriteOptions { - await_durable: true, - ..Default::default() - }, - ) - .await + let handle = db + .put_with_options(key, val, &PutOptions::default(), &WriteOptions::default()) + .await?; + handle.await_durable().await?; + Ok(handle) } // open db1 and assert that it can write. @@ -6659,13 +7066,12 @@ mod tests { b"w1", b"value", &PutOptions::default(), - &WriteOptions { - await_durable: true, - ..Default::default() - }, + &WriteOptions::default(), ) .await; - assert!(result.is_err()); + if let Ok(handle) = result { + assert!(handle.await_durable().await.is_err()); + } } async fn wait_for_wal_sst_count(table_store: &TableStore, min_count: usize, context: &str) { @@ -6721,12 +7127,12 @@ mod tests { ) .await; - let get_arrivals_before = gated_store.get_opts_gate.arrivals(); - gated_store.get_opts_gate.close(); + let head_arrivals_before = gated_store.head_gate.arrivals(); + gated_store.head_gate.close(); fail_parallel::cfg(fp_registry.clone(), "replay-wal-pause", "off").unwrap(); gated_store - .get_opts_gate - .wait_for_arrivals(get_arrivals_before + 1) + .head_gate + .wait_for_arrivals(head_arrivals_before + 1) .await; let db2 = Db::builder(path, base_store.clone()) @@ -6744,7 +7150,7 @@ mod tests { .delete_sst(&SsTableId::Wal(1)) .await .unwrap(); - gated_store.get_opts_gate.release(); + gated_store.head_gate.release(); let err = match w1_handle.await.unwrap() { Ok(_) => panic!("expected W1 open to fail"), @@ -6800,19 +7206,19 @@ mod tests { ) .await; - let get_arrivals_before = gated_store.get_opts_gate.arrivals(); - gated_store.get_opts_gate.close(); + let head_arrivals_before = gated_store.head_gate.arrivals(); + gated_store.head_gate.close(); fail_parallel::cfg(fp_registry.clone(), "replay-wal-pause", "off").unwrap(); gated_store - .get_opts_gate - .wait_for_arrivals(get_arrivals_before + 1) + .head_gate + .wait_for_arrivals(head_arrivals_before + 1) .await; probe_table_store .delete_sst(&SsTableId::Wal(1)) .await .unwrap(); - gated_store.get_opts_gate.release(); + gated_store.head_gate.release(); let err = match w1_handle.await.unwrap() { Ok(_) => panic!("expected W1 open to fail"), @@ -6859,7 +7265,8 @@ mod tests { b"1", &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() + await_durable: false, + ..Default::default() }, ) .await @@ -6912,7 +7319,8 @@ mod tests { b"1", &PutOptions::default(), &WriteOptions { - await_durable: false, ..Default::default() + await_durable: false, + ..Default::default() }, ) .await @@ -7209,8 +7617,14 @@ mod tests { kv_store.close().await.unwrap(); } + /// https://github.com/slatedb/slatedb/issues/2055: manifests are now + /// written as V2 universally (RFC-0004 Phase 2), which drops + /// `wal_object_store_uri` entirely (PR #1473). So a DB created with + /// `with_wal_object_store()` can be reopened without it configured; + /// the WAL-store reconfiguration check is a no-op once the persisted + /// manifest is V2. #[tokio::test] - async fn test_wal_store_reconfiguration_fails() { + async fn test_wal_store_reconfiguration_allowed_after_v2_manifest() { let object_store = Arc::new(InMemory::new()); let wal_object_store = Arc::new(InMemory::new()); @@ -7222,24 +7636,12 @@ mod tests { .unwrap(); kv_store.close().await.unwrap(); - let result = Db::builder("/tmp/test_kv_store", object_store) + let reopened = Db::builder("/tmp/test_kv_store", object_store) .with_settings(test_db_options(0, 1024, None)) .build() - .await; - match result { - Err(err) => { - assert!(err.to_string().contains("unsupported")); - } - _ => panic!("expected Unsupported error"), - } - } - - #[test] - fn test_write_option_defaults() { - // This is a regression test for a bug where the defaults for WriteOptions were not being - // set correctly due to visibility issues. - let write_options = WriteOptions::default(); - assert!(write_options.await_durable); + .await + .unwrap(); + reopened.close().await.unwrap(); } #[tokio::test] @@ -7331,7 +7733,7 @@ mod tests { min_filter_keys: u32, l0_sst_size_bytes: usize, compactor_options: Option, - ttl: Option, + default_ttl_millis: Option, ) -> Settings { Settings { flush_interval: Some(Duration::from_millis(100)), @@ -7352,7 +7754,7 @@ mod tests { object_store_cache_options: ObjectStoreCacheOptions::default(), garbage_collector_options: None, metric_level: MetricLevel::default(), - default_ttl: ttl, + default_ttl_millis, max_transaction_conflict_metadata_bytes: crate::config::DEFAULT_MAX_TRANSACTION_CONFLICT_METADATA_BYTES, max_retained_conflict_metadata_bytes: @@ -7506,9 +7908,7 @@ mod tests { // freeze the job after its first output SSTs upload. Manifest and // `.compactions` I/O use the ungated store passed to `Db::builder`, // so the worker's heartbeats keep flowing while the job is frozen. - let gated = Arc::new(crate::test_utils::GatedObjectStore::new( - object_store.clone(), - )); + let gated = Arc::new(GatedObjectStore::new(object_store.clone())); let gated_store: Arc = gated.clone(); let db = Db::builder(path, object_store.clone()) @@ -7672,19 +8072,13 @@ mod tests { } db.put(b"key1", b"value1").await.unwrap(); - db.inner - .flush_memtables(crate::memtable_flusher::FlushTarget::All) - .await - .unwrap(); + db.inner.flush_memtables(FlushTarget::All).await.unwrap(); let txn = db.begin(IsolationLevel::Snapshot).await.unwrap(); let txn_seq = txn.seqnum(); db.put(b"key2", b"value2").await.unwrap(); - db.inner - .flush_memtables(crate::memtable_flusher::FlushTarget::All) - .await - .unwrap(); + db.inner.flush_memtables(FlushTarget::All).await.unwrap(); let min_active_seq = db.inner.txn_manager.min_active_seq(); assert_eq!(min_active_seq, Some(txn_seq)); @@ -7701,10 +8095,7 @@ mod tests { drop(txn); db.put(b"key3", b"value3").await.unwrap(); - db.inner - .flush_memtables(crate::memtable_flusher::FlushTarget::All) - .await - .unwrap(); + db.inner.flush_memtables(FlushTarget::All).await.unwrap(); { let state = db.inner.state.read(); @@ -7725,19 +8116,13 @@ mod tests { let db = Db::builder(path, object_store).build().await.unwrap(); db.put(b"key1", b"value1").await.unwrap(); - db.inner - .flush_memtables(crate::memtable_flusher::FlushTarget::All) - .await - .unwrap(); + db.inner.flush_memtables(FlushTarget::All).await.unwrap(); let snapshot = db.snapshot().await.unwrap(); let snapshot_seq = snapshot.seq(); db.put(b"key2", b"value2").await.unwrap(); - db.inner - .flush_memtables(crate::memtable_flusher::FlushTarget::All) - .await - .unwrap(); + db.inner.flush_memtables(FlushTarget::All).await.unwrap(); let txn = db.begin(IsolationLevel::Snapshot).await.unwrap(); let txn_seq = txn.seqnum(); @@ -7750,10 +8135,7 @@ mod tests { assert!(snapshot_seq < txn_seq); db.put(b"key3", b"value3").await.unwrap(); - db.inner - .flush_memtables(crate::memtable_flusher::FlushTarget::All) - .await - .unwrap(); + db.inner.flush_memtables(FlushTarget::All).await.unwrap(); { let state = db.inner.state.read(); @@ -7768,10 +8150,7 @@ mod tests { drop(snapshot); db.put(b"key4", b"value4").await.unwrap(); - db.inner - .flush_memtables(crate::memtable_flusher::FlushTarget::All) - .await - .unwrap(); + db.inner.flush_memtables(FlushTarget::All).await.unwrap(); assert_eq!(db.inner.snapshot_manager.min_active_seq(), None); assert_eq!(db.inner.txn_manager.min_active_seq(), Some(txn_seq)); @@ -7795,19 +8174,13 @@ mod tests { let db = Db::builder(path, object_store).build().await.unwrap(); db.put(b"key1", b"value1").await.unwrap(); - db.inner - .flush_memtables(crate::memtable_flusher::FlushTarget::All) - .await - .unwrap(); + db.inner.flush_memtables(FlushTarget::All).await.unwrap(); let txn = db.begin(IsolationLevel::Snapshot).await.unwrap(); let txn_seq = txn.seqnum(); db.put(b"key2", b"value2").await.unwrap(); - db.inner - .flush_memtables(crate::memtable_flusher::FlushTarget::All) - .await - .unwrap(); + db.inner.flush_memtables(FlushTarget::All).await.unwrap(); let snapshot = db.snapshot().await.unwrap(); let snapshot_seq = snapshot.seq(); @@ -7820,10 +8193,7 @@ mod tests { assert!(txn_seq < snapshot_seq); db.put(b"key3", b"value3").await.unwrap(); - db.inner - .flush_memtables(crate::memtable_flusher::FlushTarget::All) - .await - .unwrap(); + db.inner.flush_memtables(FlushTarget::All).await.unwrap(); { let state = db.inner.state.read(); @@ -7838,10 +8208,7 @@ mod tests { drop(txn); db.put(b"key4", b"value4").await.unwrap(); - db.inner - .flush_memtables(crate::memtable_flusher::FlushTarget::All) - .await - .unwrap(); + db.inner.flush_memtables(FlushTarget::All).await.unwrap(); assert_eq!(db.inner.txn_manager.min_active_seq(), None); assert_eq!( @@ -8287,7 +8654,7 @@ mod tests { // Put with options (TTL) clock.set(200); let put_opts = PutOptions { - ttl: Ttl::ExpireAfter(1000), + ttl: Ttl::ExpireAfterMillis(1000), }; let handle = db .put_with_options( @@ -9163,7 +9530,7 @@ mod tests { // When: the DB is fenced (simulated via closed_result) db.inner .status_manager - .write_result(Err(crate::error::SlateDBError::Fenced)); + .write_result(Err(SlateDBError::Fenced)); // Then: the watcher should report close_reason = Fenced let status = tokio::time::timeout( @@ -9365,7 +9732,7 @@ mod tests { .tree .compacted .first() - .is_some_and(|sr| sr.sst_views.len() > 1) + .is_some_and(|sr| sr.sst_views().len() > 1) { break; } @@ -9523,7 +9890,7 @@ mod tests { .tree .compacted .first() - .is_some_and(|sr| sr.sst_views.len() > 1) + .is_some_and(|sr| sr.sst_views().len() > 1) { break; } @@ -9569,7 +9936,7 @@ mod tests { key, value, &PutOptions { - ttl: Ttl::ExpireAfter(50), + ttl: Ttl::ExpireAfterMillis(50), }, &WriteOptions { await_durable: false, @@ -9602,7 +9969,7 @@ mod tests { .unwrap(); let put_opts = PutOptions { - ttl: Ttl::ExpireAfter(50), + ttl: Ttl::ExpireAfterMillis(50), }; let write_opts = WriteOptions { await_durable: false, @@ -9672,13 +10039,13 @@ mod tests { .await .unwrap(); - // when: write with ExpireAt at different clock times + // when: write with ExpireAtMillis at different clock times clock.set(100); db.put_with_options( b"key1", b"value1", &PutOptions { - ttl: Ttl::ExpireAt(500), + ttl: Ttl::ExpireAtMillis(500), }, &WriteOptions { await_durable: false, @@ -9693,7 +10060,7 @@ mod tests { b"key2", b"value2", &PutOptions { - ttl: Ttl::ExpireAt(500), + ttl: Ttl::ExpireAtMillis(500), }, &WriteOptions { await_durable: false, @@ -9862,14 +10229,14 @@ mod tests { b"key1", b"a", &MergeOptions { - ttl: Ttl::ExpireAfter(3600), + ttl: Ttl::ExpireAfterMillis(3600), }, ); batch.merge_with_options( b"key1", b"b", &MergeOptions { - ttl: Ttl::ExpireAfter(7200), + ttl: Ttl::ExpireAfterMillis(7200), }, ); @@ -10010,7 +10377,7 @@ mod tests { // then: let estimated = lookup_metric( &metrics_recorder, - crate::wal_buffer::stats::WAL_BUFFER_ESTIMATED_BYTES, + crate::wal_buffer_stats::WAL_BUFFER_ESTIMATED_BYTES, ); assert!( estimated.is_some_and(|v| v > 0), @@ -10088,6 +10455,130 @@ mod tests { db.close().await.unwrap(); } + struct GaugeBlockControl { + target: &'static str, + armed: AtomicBool, + tripped: AtomicBool, + release: AtomicBool, + entered_tx: tokio::sync::mpsc::UnboundedSender<()>, + } + + struct BlockableGauge { + name: String, + control: Arc, + } + + impl GaugeFn for BlockableGauge { + fn set(&self, _value: i64) { + if self.name != self.control.target { + return; + } + if !self.control.armed.load(Ordering::SeqCst) { + return; + } + if self.control.tripped.swap(true, Ordering::SeqCst) { + return; + } + let _ = self.control.entered_tx.send(()); + while !self.control.release.load(Ordering::SeqCst) { + std::thread::sleep(Duration::from_millis(5)); + } + } + } + + struct BlockingGaugeRecorder { + control: Arc, + } + + impl MetricsRecorder for BlockingGaugeRecorder { + fn register_counter( + &self, + name: &str, + description: &str, + labels: &[(&str, &str)], + ) -> Arc { + NoopMetricsRecorder.register_counter(name, description, labels) + } + + fn register_gauge( + &self, + name: &str, + _description: &str, + _labels: &[(&str, &str)], + ) -> Arc { + Arc::new(BlockableGauge { + name: name.to_string(), + control: self.control.clone(), + }) + } + + fn register_up_down_counter( + &self, + name: &str, + description: &str, + labels: &[(&str, &str)], + ) -> Arc { + NoopMetricsRecorder.register_up_down_counter(name, description, labels) + } + + fn register_histogram( + &self, + name: &str, + description: &str, + labels: &[(&str, &str)], + boundaries: &[f64], + ) -> Arc { + NoopMetricsRecorder.register_histogram(name, description, labels, boundaries) + } + } + + #[tokio::test(flavor = "multi_thread", worker_threads = 4)] + async fn test_should_not_hold_state_lock_during_metrics_callbacks() { + // User metrics callbacks must not run under the `db.state` write + // lock. Block a gauge callback and assert `get()` still completes. + let object_store: Arc = Arc::new(InMemory::new()); + let (entered_tx, mut entered_rx) = tokio::sync::mpsc::unbounded_channel(); + let control = Arc::new(GaugeBlockControl { + target: crate::db_stats::L0_SST_COUNT, + armed: AtomicBool::new(false), + tripped: AtomicBool::new(false), + release: AtomicBool::new(false), + entered_tx, + }); + let recorder = Arc::new(BlockingGaugeRecorder { + control: control.clone(), + }); + let db = Db::builder( + "/tmp/test_should_not_hold_state_lock_during_metrics_callbacks", + object_store, + ) + .with_settings(test_db_options(0, 1024, None)) + .with_metrics_recorder(recorder) + .build() + .await + .unwrap(); + + control.armed.store(true, Ordering::SeqCst); + tokio::time::timeout(Duration::from_secs(10), entered_rx.recv()) + .await + .expect("manifest stats gauge was never set") + .unwrap(); + + let read_task = tokio::spawn({ + let db = db.clone(); + async move { db.get(b"key").await } + }); + let read_result = tokio::time::timeout(Duration::from_secs(5), read_task).await; + control.release.store(true, Ordering::SeqCst); + let value = read_result + .expect("db.get() deadlocked while a metrics callback was in flight") + .unwrap() + .unwrap(); + assert_eq!(value, None); + + db.close().await.unwrap(); + } + #[tokio::test] async fn test_should_record_segment_max_l0_sst_count_with_extractor() { // With a segment extractor configured, `l0_sst_count` sums L0 SSTs @@ -10390,6 +10881,12 @@ mod tests { .build() .await .unwrap(); + let wal_store = crate::wal::slatedb::store::WalTableStore::new( + object_store.clone(), + SsTableFormat::default(), + path, + TableStoreKind::Main, + ); let mut batch = WriteBatch::new(); for id in 0..64 { batch.put(format!("oversized-{id:03}"), vec![id as u8; 2048]); @@ -10406,9 +10903,7 @@ mod tests { .unwrap() .last_flushed_wal_id; assert!(last_flushed_wal_id > 0); - let listed = source - .inner - .table_store + let listed = wal_store .list_wal_ssts_for_replay(1..last_flushed_wal_id + 1) .await .unwrap(); @@ -10499,7 +10994,7 @@ mod tests { async fn test_wal_replay_rejects_empty_extractor_prefix() { #[derive(Debug)] struct AliasedAlwaysEmptyExtractor; - impl crate::PrefixExtractor for AliasedAlwaysEmptyExtractor { + impl PrefixExtractor for AliasedAlwaysEmptyExtractor { fn name(&self) -> &str { "fixed-3" } @@ -10722,10 +11217,10 @@ mod tests { } impl ExtractorConfig { - fn to_extractor(self) -> Option> { + fn to_extractor(self) -> Option> { #[derive(Debug)] struct OtherExtractor; - impl crate::PrefixExtractor for OtherExtractor { + impl PrefixExtractor for OtherExtractor { fn name(&self) -> &str { "other" } @@ -11704,12 +12199,12 @@ mod tests { } /// A path under the db root, e.g. `sub_path("wal/00..002.sst")`. - fn sub_path(&self, suffix: &str) -> object_store::path::Path { - object_store::path::Path::from(format!("{}/{}", self.db_path, suffix)) + fn sub_path(&self, suffix: &str) -> Path { + Path::from(format!("{}/{}", self.db_path, suffix)) } /// Number of cached part files for an object. - fn cached_part_count(&self, path: &object_store::path::Path) -> usize { + fn cached_part_count(&self, path: &Path) -> usize { let dir = self.cache_root.join(path.to_string()); let Ok(entries) = std::fs::read_dir(dir) else { return 0; @@ -11725,7 +12220,7 @@ mod tests { .count() } - fn assert_cached(&self, path: &object_store::path::Path, expected_parts: usize) { + fn assert_cached(&self, path: &Path, expected_parts: usize) { assert_eq!( self.cached_part_count(path), expected_parts, @@ -11746,7 +12241,7 @@ mod tests { } /// Lists the compacted SSTs currently in the object store. - async fn compacted_locations(&self) -> Vec { + async fn compacted_locations(&self) -> Vec { let prefix = self.sub_path("compacted"); self.upstream .list(Some(&prefix)) @@ -11756,13 +12251,13 @@ mod tests { } /// The size of an object as stored upstream, in bytes. - async fn object_size(&self, path: &object_store::path::Path) -> u64 { + async fn object_size(&self, path: &Path) -> u64 { self.upstream.head(path).await.unwrap().size } /// The upstream path of a compacted SST id. - fn compacted_sst_path(&self, id: &SsTableId) -> object_store::path::Path { - crate::paths::PathResolver::from_root(self.db_path.as_str()).sst_path(id) + fn compacted_sst_path(&self, id: &SsTableId) -> Path { + PathResolver::from_root(self.db_path.as_str()).sst_path(id) } fn l0_ids(&self) -> Vec { @@ -11795,7 +12290,7 @@ mod tests { .manifest() .compacted() .iter() - .flat_map(|sr| sr.sst_views.iter()) + .flat_map(|sr| sr.sst_views().iter()) .map(|v| v.sst.id) .filter(|id| !l0_ids.contains(id)) .collect() @@ -11913,26 +12408,56 @@ mod tests { .await .unwrap(); - let requests_before = - lookup_object_store_op_request_count(&metrics_recorder, "db", "main", "get"); + let requests_before = lookup_object_store_api_request_count( + &metrics_recorder, + "db", + "main", + "get", + "get_range", + ); + let histograms_before = lookup_object_store_api_histogram_count( + &metrics_recorder, + "db", + "main", + "get", + "get_range", + ); let _val = kv_store.get(b"test_key").await.unwrap(); - let requests_after_first = - lookup_object_store_op_request_count(&metrics_recorder, "db", "main", "get"); + let requests_after_first = lookup_object_store_api_request_count( + &metrics_recorder, + "db", + "main", + "get", + "get_range", + ); let got = kv_store.get(b"test_key").await.unwrap(); - let requests_after_second = - lookup_object_store_op_request_count(&metrics_recorder, "db", "main", "get"); + let requests_after_second = lookup_object_store_api_request_count( + &metrics_recorder, + "db", + "main", + "get", + "get_range", + ); + let histograms_after_second = lookup_object_store_api_histogram_count( + &metrics_recorder, + "db", + "main", + "get", + "get_range", + ); // The instrumented store sits above the object store cache and counts // logical read calls whether they are served from the cache or the - // remote store. + // remote store. Restrict the assertion to range reads so background + // manifest polling, which uses plain get calls, cannot affect it. // A point get reads the single-part SST in three sub-ranges (index, // filter and block). assert_eq!(requests_after_first, requests_before + 3); assert_eq!(got, Some(Bytes::from_static(b"test_value"))); assert_eq!(requests_after_second, requests_after_first + 3); assert_eq!( - lookup_object_store_op_histogram_count(&metrics_recorder, "db", "main", "get"), - requests_after_second as u64 + histograms_after_second - histograms_before, + (requests_after_second - requests_before) as u64 ); kv_store.close().await.unwrap(); } diff --git a/slatedb/src/db/builder.rs b/slatedb/src/db/builder.rs index aadc184193..1ea322eca2 100644 --- a/slatedb/src/db/builder.rs +++ b/slatedb/src/db/builder.rs @@ -162,8 +162,11 @@ use crate::retrying_object_store::RetryingObjectStore; use crate::tablestore::{TableStore, TableStoreKind}; use crate::utils::SafeSender; use crate::utils::WatchableOnceCell; +use crate::wal; +use crate::wal::slatedb::admin::SlateDbWalAdmin; +use crate::wal::slatedb::store::WalTableStore; use crate::wal::wal_disabled::DisabledWalObserver; -use crate::wal::WalObserver; +use crate::wal::{WalAdmin, WalGc, WalObserver}; use slatedb_common::clock::DefaultSystemClock; use slatedb_common::clock::SystemClock; use slatedb_common::metrics::MetricsRecorder; @@ -195,6 +198,7 @@ pub struct DbBuilder> { filter_policies: Vec>, metrics_recorder: Arc, segment_extractor: Option>, + wal_writer_init: Option>, writer_epoch: Option, } @@ -225,6 +229,7 @@ impl> DbBuilder

{ filter_policies: default_filter_policies(), metrics_recorder: Arc::new(NoopMetricsRecorder::new()), segment_extractor: None, + wal_writer_init: None, writer_epoch: None, } } @@ -279,6 +284,14 @@ impl> DbBuilder

{ self } + /// Sets the WAL writer initializer used to fence the WAL and create the writer. + /// Use this to plug in a custom WAL implementation. SlateDB's object-store-backed + /// WAL is used by default. + pub fn with_wal_writer(mut self, wal_writer_init: Box) -> Self { + self.wal_writer_init = Some(wal_writer_init); + self + } + /// Sets the cache to use for the database for caching sst blocks /// /// SlateDB uses a cache to efficiently store and retrieve blocks and SST metadata locally. @@ -512,7 +525,11 @@ impl> DbBuilder

{ ); let retrying_wal_object_store: Option> = self .wal_object_store + .clone() .map(|s| wrap_object_store(s, ObjectStoreComponent::Db, ObjectStoreType::Wal)); + let wal_object_store = retrying_wal_object_store + .clone() + .unwrap_or_else(|| retrying_main_object_store.clone()); // Log the database opening if let Ok(settings_json) = self.settings.to_json_string() { @@ -619,6 +636,13 @@ impl> DbBuilder

{ TableStoreKind::Main, self.block_cache_policy.clone(), )); + let wal_store = Arc::new(WalTableStore::new_with_fp_registry( + wal_object_store, + sst_format.clone(), + path_resolver.clone(), + self.fp_registry.clone(), + TableStoreKind::Main, + )); // Initialize the database let stored_manifest = match latest_manifest { @@ -653,22 +677,20 @@ impl> DbBuilder

{ let fencer = WriterFencer::new( status_manager.result_reader(), recorder.clone(), - table_store.clone(), + wal_store.clone(), &self.settings, system_clock.clone(), task_executor.clone(), + self.wal_writer_init, ); let stage_started = system_clock.now(); let WriterFenceResult { manifest, - replay_range, + replay_iterator, mut wal_writer, } = fencer.fence(stored_manifest, self.writer_epoch).await?; - let replay_start_wal_id = replay_range.start; - let replay_end_wal_id = replay_range.end; - let replay_wal_count = replay_end_wal_id.saturating_sub(replay_start_wal_id); info!( - "SlateDB writer open stage completed [stage=writer_fence, elapsed_ms={}, total_elapsed_ms={}, replay_start_wal_id={}, replay_end_wal_id={}, replay_wal_count={}]", + "SlateDB writer open stage completed [stage=writer_fence, elapsed_ms={}, total_elapsed_ms={}]", system_clock .now() .signed_duration_since(stage_started) @@ -676,10 +698,7 @@ impl> DbBuilder

{ system_clock .now() .signed_duration_since(open_started) - .num_milliseconds(), - replay_start_wal_id, - replay_end_wal_id, - replay_wal_count + .num_milliseconds() ); let (wal_writer, wal_observer) = if DbInner::wal_enabled_in_options(&self.settings) { let wal_observer = wal_writer.observer(); @@ -892,13 +911,10 @@ impl> DbBuilder

{ // Replay WAL let stage_started = system_clock.now(); + info!("SlateDB WAL replay started"); + inner.replay_wal(replay_iterator).await?; info!( - "SlateDB WAL replay started [replay_start_wal_id={}, replay_end_wal_id={}, replay_wal_count={}]", - replay_start_wal_id, replay_end_wal_id, replay_wal_count - ); - inner.replay_wal(replay_range).await?; - info!( - "SlateDB writer open stage completed [stage=wal_replay, elapsed_ms={}, total_elapsed_ms={}, replay_start_wal_id={}, replay_end_wal_id={}, replay_wal_count={}]", + "SlateDB writer open stage completed [stage=wal_replay, elapsed_ms={}, total_elapsed_ms={}]", system_clock .now() .signed_duration_since(stage_started) @@ -906,10 +922,7 @@ impl> DbBuilder

{ system_clock .now() .signed_duration_since(open_started) - .num_milliseconds(), - replay_start_wal_id, - replay_end_wal_id, - replay_wal_count + .num_milliseconds() ); // Preload cache if enabled @@ -954,6 +967,7 @@ pub struct AdminBuilder> { path: P, main_object_store: Arc, wal_object_store: Option>, + wal_admin: Option>, system_clock: Arc, rand: Arc, object_store_max_retries: Option, @@ -969,6 +983,7 @@ impl> AdminBuilder

{ path, main_object_store, wal_object_store: None, + wal_admin: None, system_clock: Arc::new(DefaultSystemClock::new()), rand: Arc::new(DbRand::default()), object_store_max_retries: None, @@ -994,6 +1009,13 @@ impl> AdminBuilder

{ self } + /// Sets the WAL admin API implementation to use for admin operations like db deletion + /// and cloning. Users should set this if using a custom WAL implementation. + pub fn with_wal_admin(mut self, wal_admin: Arc) -> Self { + self.wal_admin = Some(wal_admin); + self + } + /// Sets the random number generator to use for randomness. pub fn with_seed(mut self, seed: u64) -> Self { self.rand = Arc::new(DbRand::new(seed)); @@ -1026,9 +1048,23 @@ impl> AdminBuilder

{ // rather than at build time, because several admin operations delegate // to sub-builders (compactor/GC) that add their own retry layer, and // wrapping here would double-wrap them. + let object_stores = ObjectStores::new(self.main_object_store, self.wal_object_store); + let wal_admin = self.wal_admin.unwrap_or_else(|| { + let retrying_object_store = Arc::new(RetryingObjectStore::new( + object_stores.store_of(ObjectStoreType::Wal).clone(), + self.rand.clone(), + self.system_clock.clone(), + self.object_store_max_retries, + )); + Arc::new(SlateDbWalAdmin::new( + retrying_object_store, + Arc::new(FailPointRegistry::new()), + )) + }); Admin { path: self.path.into(), - object_stores: ObjectStores::new(self.main_object_store, self.wal_object_store), + object_stores, + wal_admin, system_clock: self.system_clock, rand: self.rand, object_store_max_retries: self.object_store_max_retries, @@ -1046,6 +1082,7 @@ pub struct GarbageCollectorBuilder> { path: P, main_object_store: Arc, wal_object_store: Option>, + wal_gc: Option>, options: GarbageCollectorOptions, gc_filter: Option>, metrics_recorder: Arc, @@ -1059,6 +1096,7 @@ impl> GarbageCollectorBuilder

{ path, main_object_store, wal_object_store: None, + wal_gc: None, options: GarbageCollectorOptions::default(), gc_filter: None, metrics_recorder: Arc::new(NoopMetricsRecorder::new()), @@ -1073,6 +1111,7 @@ impl> GarbageCollectorBuilder

{ path: self.path.into(), main_object_store: self.main_object_store, wal_object_store: self.wal_object_store, + wal_gc: self.wal_gc, options: self.options, gc_filter: self.gc_filter, metrics_recorder: self.metrics_recorder, @@ -1120,6 +1159,11 @@ impl> GarbageCollectorBuilder

{ self } + pub fn with_wal_gc(mut self, wal_gc: Arc) -> Self { + self.wal_gc = Some(wal_gc); + self + } + /// Builds a GarbageCollector using pre-existing stores (used by DbBuilder). pub(crate) fn build_collector( self, @@ -1141,6 +1185,7 @@ impl> GarbageCollectorBuilder

{ &recorder, self.system_clock, self.gc_filter, + self.wal_gc, ) } @@ -1199,6 +1244,7 @@ impl> GarbageCollectorBuilder

{ &recorder, self.system_clock, self.gc_filter, + self.wal_gc, ) } } @@ -1214,7 +1260,7 @@ pub(crate) struct CompactorHandlers { /// The embedded worker handler and its receiver, present when /// [`CompactorOptions::worker`] is `Some`. pub(crate) worker: Option<( - crate::compaction_worker::CompactionWorkerHandler, + CompactionWorkerHandler, async_channel::Receiver, )>, } @@ -1728,6 +1774,7 @@ pub struct DbReaderBuilder> { path: P, object_store: Arc, wal_object_store: Option>, + wal_reader: Option>, db_cache: Option>, mode: DbReaderMode, merge_operator: Option, @@ -1747,6 +1794,7 @@ impl> DbReaderBuilder

{ path, object_store, wal_object_store: None, + wal_reader: None, db_cache: default_db_cache(), mode: DbReaderMode::default(), merge_operator: None, @@ -1773,6 +1821,13 @@ impl> DbReaderBuilder

{ self } + /// Sets the WAL reader used to replay WAL rows. Use this to plug in a custom + /// WAL implementation. SlateDB's object-store-backed WAL is used by default. + pub fn with_wal_reader(mut self, wal_reader: Arc) -> Self { + self.wal_reader = Some(wal_reader); + self + } + /// Sets the merge operator to use when reading merge operands. pub fn with_merge_operator(mut self, merge_operator: MergeOperatorType) -> Self { self.merge_operator = Some(merge_operator); @@ -1895,7 +1950,7 @@ impl> DbReaderBuilder

{ .await?; let maybe_cached_object_store: Arc = match &maybe_cached { Some(cached) => Arc::clone(cached) as Arc, - None => self.object_store, + None => self.object_store.clone(), }; let retrying_object_store = instrumented_retrying_object_store( @@ -1908,18 +1963,20 @@ impl> DbReaderBuilder

{ self.options.object_store_max_retries, ); - let retrying_wal_object_store: Option> = - self.wal_object_store.map(|s| { - instrumented_retrying_object_store( - s, - &recorder, - ObjectStoreComponent::Reader, - ObjectStoreType::Wal, - self.rand.clone(), - self.system_clock.clone(), - self.options.object_store_max_retries, - ) - }); + let retrying_wal_object_store = self.wal_object_store.map(|wal_object_store| { + instrumented_retrying_object_store( + wal_object_store, + &recorder, + ObjectStoreComponent::Reader, + ObjectStoreType::Wal, + self.rand.clone(), + self.system_clock.clone(), + self.options.object_store_max_retries, + ) + }); + let wal_object_store = retrying_wal_object_store + .clone() + .unwrap_or_else(|| retrying_object_store.clone()); // Validate WAL object store configuration. let manifest_store = Arc::new(ManifestStore::new(&path, retrying_object_store.clone())); @@ -1968,20 +2025,30 @@ impl> DbReaderBuilder

{ .wal_replay .validate_wal_block_size(sst_format.block_size)?; let path_resolver = PathResolver::new_with_external_ssts(path.clone(), external_ssts); + let fp_registry = Arc::new(FailPointRegistry::new()); let table_store = Arc::new(TableStore::new_with_fp_registry( ObjectStores::new(retrying_object_store, retrying_wal_object_store), - sst_format, - path_resolver, - Arc::new(FailPointRegistry::new()), + sst_format.clone(), + path_resolver.clone(), + Arc::clone(&fp_registry), wrapped_cache, TableStoreKind::Reader, BlockCachePolicy::default(), )); + let wal_store = Arc::new(WalTableStore::new_with_fp_registry( + wal_object_store, + sst_format, + path_resolver, + fp_registry, + TableStoreKind::Reader, + )); let mut reader = DbReader::open_internal( manifest_store, table_store, + wal_store, self.mode, + self.wal_reader, self.merge_operator, self.segment_extractor, self.options, @@ -2070,6 +2137,7 @@ pub struct CloneBuilder + Clone = (Bound, Bound>, object_store: Arc, wal_object_store: Option>, + wal_admin: Option>, system_clock: Option>, rand: Option>, projection_range: Option, @@ -2088,6 +2156,7 @@ impl + Clone> CloneBuilder { sources: vec![source], object_store, wal_object_store: None, + wal_admin: None, system_clock: None, rand: None, projection_range: None, @@ -2116,6 +2185,11 @@ impl + Clone> CloneBuilder { self } + pub fn with_wal_admin(mut self, wal_admin: Arc) -> Self { + self.wal_admin = Some(wal_admin); + self + } + pub fn with_projection_range(mut self, projection_range: Option) -> Self { self.projection_range = projection_range; self @@ -2150,7 +2224,7 @@ impl + Clone> CloneBuilder { r.start_bound().cloned(), r.end_bound().cloned(), ) - .ok_or_else(|| crate::error::SlateDBError::InvalidProjection { + .ok_or_else(|| SlateDBError::InvalidProjection { prefix: Bytes::copy_from_slice(prefix), reason: "empty range".into(), }) @@ -2170,11 +2244,21 @@ impl + Clone> CloneBuilder { /// Build and execute the clone operation. pub async fn build(self) -> Result<(), crate::Error> { + let fp_registry = Arc::new(FailPointRegistry::new()); + let wal_admin = if let Some(wal_admin) = self.wal_admin { + wal_admin + } else { + Arc::new(SlateDbWalAdmin::new( + self.wal_object_store.unwrap_or(self.object_store.clone()), + fp_registry.clone(), + )) + }; crate::clone::create_clone( self.sources, self.clone_path, - ObjectStores::new(self.object_store, self.wal_object_store), - Arc::new(FailPointRegistry::new()), + self.object_store, + wal_admin, + fp_registry, self.system_clock .unwrap_or_else(|| Arc::new(DefaultSystemClock::new())), self.rand.unwrap_or_else(|| Arc::new(Default::default())), @@ -2268,6 +2352,10 @@ mod tests { use crate::instrumented_object_store::stats::REQUEST_COUNT as OBJECT_STORE_REQUEST_COUNT; use crate::manifest::store::{ManifestStore, StoredManifest}; use crate::manifest::ManifestCore; + use crate::wal::test_utils::FakeWalWriter; + use crate::wal::{ + WalError, WalIterator, WalRows, WriterInit, WriterInitResult, WriterManifest, + }; use object_store::memory::InMemory; use object_store::path::Path; use object_store::ObjectStore; @@ -2275,7 +2363,37 @@ mod tests { use slatedb_common::metrics::{ lookup_metric, lookup_metric_with_labels, DefaultMetricsRecorder, MetricsRecorderHelper, }; - use std::sync::Arc; + use std::sync::{ + atomic::{AtomicBool, Ordering}, + Arc, + }; + + struct EmptyWalIterator; + + #[async_trait::async_trait] + impl WalIterator for EmptyWalIterator { + async fn next(&mut self) -> Result, WalError> { + Ok(None) + } + } + + struct RecordingWriterInit { + called: Arc, + } + + #[async_trait::async_trait] + impl WriterInit for RecordingWriterInit { + async fn fence_and_init( + &self, + manifest: &mut WriterManifest, + ) -> Result { + self.called.store(true, Ordering::Relaxed); + Ok(WriterInitResult { + replay_iterator: Box::new(EmptyWalIterator), + wal_writer: Box::new(FakeWalWriter::new(manifest.replay_after_wal_id())), + }) + } + } fn object_store_labels( component: &'static str, @@ -2301,6 +2419,29 @@ mod tests { assert_eq!(lookup_metric(recorder, name), Some(1)); } + #[tokio::test] + async fn test_db_builder_uses_custom_wal_writer_init() { + let called = Arc::new(AtomicBool::new(false)); + let db = crate::Db::builder( + "test_db_builder_uses_custom_wal_writer_init", + Arc::new(InMemory::new()), + ) + .with_settings(Settings { + compactor_options: None, + garbage_collector_options: None, + ..Settings::default() + }) + .with_wal_writer(Box::new(RecordingWriterInit { + called: Arc::clone(&called), + })) + .build() + .await + .expect("failed to build db"); + + assert!(called.load(Ordering::Relaxed)); + db.close().await.expect("failed to close db"); + } + #[tokio::test] async fn test_db_builder_starts_gc_by_default() { let metrics_recorder = Arc::new(DefaultMetricsRecorder::new()); diff --git a/slatedb/src/db_cache/mod.rs b/slatedb/src/db_cache/mod.rs index dabde9786a..972e9757e3 100644 --- a/slatedb/src/db_cache/mod.rs +++ b/slatedb/src/db_cache/mod.rs @@ -423,7 +423,7 @@ impl From<(SsTableId, u64)> for CachedKey { #[derive(Clone)] pub(crate) struct EncodedCachedFilter { pub(crate) name: String, - pub(crate) data: bytes::Bytes, + pub(crate) data: Bytes, } #[non_exhaustive] @@ -1688,11 +1688,11 @@ mod tests { // given: a cache that always returns errors let recorder = Arc::new(DefaultMetricsRecorder::new()); let helper = MetricsRecorderHelper::new(recorder.clone(), MetricLevel::default()); - let failing_cache: Arc = Arc::new(super::test_utils::FailingCache); - let cache = super::DbCacheWrapper::new( + let failing_cache: Arc = Arc::new(super::test_utils::FailingCache); + let cache = DbCacheWrapper::new( failing_cache, &helper, - Arc::new(slatedb_common::clock::DefaultSystemClock::default()), + Arc::new(DefaultSystemClock::default()), ); let key = CachedKey::from((SST_ID, 12345u64)); diff --git a/slatedb/src/db_cache/serde.rs b/slatedb/src/db_cache/serde.rs index bae181285d..a435418480 100644 --- a/slatedb/src/db_cache/serde.rs +++ b/slatedb/src/db_cache/serde.rs @@ -321,9 +321,9 @@ mod tests { let policy = BloomFilterPolicy::new(10); let mut builder = policy.builder(); for k in [b"foo", b"bar", b"baz"] { - builder.add_entry(&crate::types::RowEntry::new( - bytes::Bytes::copy_from_slice(k), - crate::types::ValueDeletable::Value(bytes::Bytes::new()), + builder.add_entry(&RowEntry::new( + Bytes::copy_from_slice(k), + crate::types::ValueDeletable::Value(Bytes::new()), 0, None, None, diff --git a/slatedb/src/db_iter.rs b/slatedb/src/db_iter.rs index 119f0a08fa..4cbe6a062f 100644 --- a/slatedb/src/db_iter.rs +++ b/slatedb/src/db_iter.rs @@ -315,9 +315,7 @@ impl DbIterator { match entry_opt { Some(entry) => { if entry.value.is_tombstone() { - return Err(crate::Error::from( - crate::error::SlateDBError::UnexpectedTombstone, - )); + return Err(crate::Error::from(SlateDBError::UnexpectedTombstone)); } Ok(Some(KeyValue::from(entry))) } @@ -331,7 +329,10 @@ impl DbIterator { Err(error) } else { let result = loop { - match self.iter.next().await { + let next = self.iter.next().await; + // Keep cached iteration cooperative. + tokio::task::coop::consume_budget().await; + match next { Ok(Some(entry)) => match entry.value { ValueDeletable::Tombstone => continue, _ => break Ok(Some(entry)), @@ -470,7 +471,10 @@ impl DbRecencyIterator { self.current_initialized = true; } - match iter.next().await { + let next = iter.next().await; + // Keep cached iteration cooperative. + tokio::task::coop::consume_budget().await; + match next { Ok(Some(entry)) => return Ok(Some(entry)), Ok(None) => { self.iters.pop_front(); diff --git a/slatedb/src/db_reader.rs b/slatedb/src/db_reader.rs index f44928a2ab..456f93ec9f 100644 --- a/slatedb/src/db_reader.rs +++ b/slatedb/src/db_reader.rs @@ -1,56 +1,58 @@ -use crate::bytes_range::{ByteRangeBounds, BytesRange}; -use crate::cached_object_store::CachedObjectStore; -use crate::clock::MonotonicClock; -use crate::config::{CheckpointOptions, DbReaderOptions, ReadOptions, ScanOptions}; -use crate::db_cache::CacheTarget; -use crate::db_cache_manager; -use crate::db_common::extract_segment_prefix; -use crate::db_iter::DbIteratorGuard; -use crate::db_state::{collect_touched_segments, SsTableId}; -use crate::db_stats::DbStats; -use crate::db_status::{ClosedResultWriter, DbStatus, DbStatusManager}; -use crate::dispatcher::{MessageHandler, MessageHandlerExecutor, MessageTickerDef, Notifier}; -use crate::error::SlateDBError; -use crate::manifest::store::{ManifestStore, StoredManifest}; -use crate::manifest::{Manifest, ManifestCore, VersionedManifest}; -use crate::mem_table::{ImmutableMemtable, KVTable, WritableKVTable}; -use crate::merge_operator::MergeOperatorType; -use crate::oracle::DbReaderOracle; -use crate::paths::PathResolver; -use crate::prefix_extractor::PrefixExtractor; -use crate::reader::{DbStateReader, Reader, ScanContext}; -use crate::replay_task_scope::ReplayTaskScope; -use crate::tablestore::TableStore; -use crate::types::KeyValue; -use crate::utils::{panic_string, IdGenerator}; -use crate::wal_replay::{ - ExactWalReplayIterator, ExactWalReplaySource, ReplayedMemtable, RuntimeWalReplayError, - RuntimeWalReplayIterator, RuntimeWalReplayOptions, RuntimeWalReplaySource, WalReplayOptions, +use crate::wal::slatedb::reader::SlateDbWalReaderOptions; +use { + crate::{ + bytes_range::{ByteRangeBounds, BytesRange}, + cached_object_store::CachedObjectStore, + clock::MonotonicClock, + config::{CheckpointOptions, DbReaderOptions, ReadOptions, ScanOptions}, + db_cache::CacheTarget, + db_cache_manager, + db_common::extract_segment_prefix, + db_iter::DbIteratorGuard, + db_state::{collect_touched_segments, SsTableId}, + db_stats::DbStats, + db_status::{ClosedResultWriter, DbStatus, DbStatusManager}, + dispatcher::{MessageHandler, MessageHandlerExecutor, MessageTickerDef}, + error::SlateDBError, + manifest::{ + store::{ManifestStore, StoredManifest}, + Manifest, ManifestCore, VersionedManifest, + }, + mem_table::{ImmutableMemtable, KVTable, WritableKVTable}, + merge_operator::MergeOperatorType, + oracle::DbReaderOracle, + paths::PathResolver, + prefix_extractor::PrefixExtractor, + reader::{DbStateReader, Reader, ScanContext}, + tablestore::TableStore, + types::KeyValue, + utils::IdGenerator, + wal::slatedb::store::WalTableStore, + wal::WalReader as WalReaderTrait, + wal_replay::{WalReplayIterator, WalReplayOptions}, + Checkpoint, DbCacheManagerOps, DbIterator, DbMetadataOps, DbReadOps, DbSnapshot, + }, + async_trait::async_trait, + bytes::Bytes, + futures::stream::BoxStream, + log::{info, warn}, + object_store::{path::Path, ObjectStore}, + parking_lot::RwLock, + slatedb_common::{clock::SystemClock, DbRand}, + std::{ + collections::{BTreeSet, HashMap}, + ops::Sub, + sync::{ + atomic::{AtomicUsize, Ordering}, + Arc, LazyLock, Weak, + }, + }, + tokio::runtime::Handle, + tokio::sync::Notify, + uuid::Uuid, }; -use crate::{Checkpoint, DbIterator, DbSnapshot}; -use crate::{DbCacheManagerOps, DbMetadataOps, DbReadOps}; -use async_trait::async_trait; -use bytes::Bytes; -use futures::{stream::BoxStream, FutureExt}; -use log::{error, info, warn}; -use object_store::path::Path; -use object_store::ObjectStore; -use parking_lot::RwLock; -use slatedb_common::clock::SystemClock; -use slatedb_common::DbRand; -use std::collections::{BTreeSet, HashMap}; -use std::ops::Sub; -use std::panic::AssertUnwindSafe; -use std::sync::atomic::{AtomicUsize, Ordering}; -use std::sync::LazyLock; -use std::sync::{Arc, Weak}; -use tokio::runtime::Handle; -use tokio::sync::Notify; -use uuid::Uuid; pub(crate) const DB_READER_TASK_NAME: &str = "manifest_poller"; -const MAX_RUNTIME_WALS_PER_TURN: u64 = 64; -const MAX_RUNTIME_REPLAY_TURN_TIME: std::time::Duration = std::time::Duration::from_secs(1); /// Determines how a [`DbReader`] chooses and refreshes the database state it reads. #[derive(Clone, Copy, Debug, Default, Eq, PartialEq)] @@ -82,6 +84,39 @@ impl From> for DbReaderMode { } } +/// Where a reader stops replaying the WAL when it builds its state. +/// +/// This is only reached when replay is wanted at all; a reader configured with +/// [`DbReaderOptions::skip_wal_replay`] reads no WAL, which +/// [`WalReplayEnd::for_reader`] expresses as `None`. +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +enum WalReplayEnd { + /// Stop at the manifest's `next_wal_sst_id`, replaying exactly the WAL that + /// the manifest itself records as durable. + Manifest, + + /// Ask the configured WAL reader for the newest WAL file and replay through it, + /// picking up writes made after the manifest was written. + Latest, +} + +impl WalReplayEnd { + /// Returns `None` when the reader is configured to skip WAL replay, in which + /// case it observes only the state recorded in the manifest (L0 and below). + fn for_reader(mode: DbReaderMode, options: &DbReaderOptions) -> Option { + if options.skip_wal_replay { + return None; + } + Some(match mode { + // A pinned checkpoint reads the state its manifest captured, so it + // stops at that manifest's WAL boundary instead of following WAL + // files written after the checkpoint was taken. + DbReaderMode::Checkpoint(_) => Self::Manifest, + DbReaderMode::ManagedCheckpoint | DbReaderMode::FollowLatest => Self::Latest, + }) + } +} + /// Read-only interface for accessing a database from either /// the latest persistent state or from an arbitrary checkpoint. /// @@ -98,6 +133,7 @@ pub struct DbReader { pub(crate) struct DbReaderInner { manifest_store: Arc, table_store: Arc, + wal_reader: Arc, options: DbReaderOptions, mode: DbReaderMode, state: RwLock>, @@ -108,8 +144,6 @@ pub(crate) struct DbReaderInner { status_manager: DbStatusManager, segment_extractor: Option>, rand: Arc, - /// Root ownership scope for every replay task in this reader tenure. - replay_tasks: ReplayTaskScope, /// Kept alive so the underlying `MetricsRecorder` is not dropped while /// metric handles in `DbStats` (and other stats structs) are still in use. /// See: https://github.com/slatedb/slatedb/issues/1469 @@ -117,58 +151,9 @@ pub(crate) struct DbReaderInner { recorder: slatedb_common::metrics::MetricsRecorderHelper, } +#[derive(Debug)] enum DbReaderMessage { PollManifest, - PollWalTail, - ContinueWalTail, - WalTailComplete(Result), - ManifestRefreshComplete(Result, SlateDBError>), -} - -impl std::fmt::Debug for DbReaderMessage { - fn fmt(&self, formatter: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - formatter.write_str(match self { - Self::PollManifest => "PollManifest", - Self::PollWalTail => "PollWalTail", - Self::ContinueWalTail => "ContinueWalTail", - Self::WalTailComplete(Ok(_)) => "WalTailComplete(Ok)", - Self::WalTailComplete(Err(_)) => "WalTailComplete(Err)", - Self::ManifestRefreshComplete(Ok(Some(_))) => "ManifestRefreshComplete(Ok(Some))", - Self::ManifestRefreshComplete(Ok(None)) => "ManifestRefreshComplete(Ok(None))", - Self::ManifestRefreshComplete(Err(_)) => "ManifestRefreshComplete(Err)", - }) - } -} - -#[derive(Clone, Copy, Debug, Eq, PartialEq)] -enum ReaderReplayPath { - ExactOpen { - include_unmanifested: bool, - source: ExactWalReplaySource, - }, - RuntimeManifest, -} - -#[derive(Clone, Copy, Debug, Eq, PartialEq)] -enum RuntimeMissingPolicy { - Error, - ExactNextIsCaughtUp(u64), -} - -struct RuntimeReplayTurn { - manifest_id: u64, - base_last_wal_id: u64, - imm_memtable: ReplayMemtables, - last_wal_id: u64, - last_committed_seq: u64, - caught_up: bool, - reached_turn_limit: bool, -} - -struct ManifestRefreshCandidate { - base_manifest_id: u64, - base_last_wal_id: u64, - state: ReaderState, } #[derive(Clone)] @@ -403,6 +388,8 @@ impl DbReaderInner { async fn new( manifest_store: Arc, table_store: Arc, + wal_store: Arc, + wal_reader: Option>, options: DbReaderOptions, mode: DbReaderMode, merge_operator: Option, @@ -422,27 +409,39 @@ impl DbReaderInner { } else { (manifest.id(), manifest.manifest().clone()) }; - let include_unmanifested = - !matches!(mode, DbReaderMode::Checkpoint(_)) && !options.skip_wal_replay; + let status_manager = DbStatusManager::new_with_initial_values( + initial_manifest.core.last_l0_seq, + VersionedManifest::from_manifest(manifest_id, initial_manifest.clone()), + BTreeSet::default(), + ); + let wal_reader = wal_reader.unwrap_or_else(|| { + Arc::new( + crate::wal::slatedb::reader::SlateDbWalReader::new_with_status_manager( + wal_store, + &status_manager, + Arc::clone(&system_clock), + SlateDbWalReaderOptions { + sst_batch_size: options.wal_replay.max_concurrent_objects, + read_ahead_bytes: options.wal_replay.block_working_memory_limit(), + ..SlateDbWalReaderOptions::default() + }, + ), + ) + }); let db_stats = DbStats::new(&recorder); - let replay_tasks = ReplayTaskScope::new(); let initial_state = Arc::new( Self::build_reader_state( checkpoint, manifest_id, initial_manifest, ReplayMemtables::default(), - ReaderReplayPath::ExactOpen { - include_unmanifested, - source: ExactWalReplaySource::ReaderOpen, - }, + WalReplayEnd::for_reader(mode, &options), Arc::clone(&table_store), + wal_reader.as_ref(), &options, segment_extractor.as_ref(), None, &db_stats, - Some(replay_tasks.clone()), - &system_clock, ) .await?, ); @@ -452,13 +451,15 @@ impl DbReaderInner { initial_state.core().last_l0_clock_tick, )); - // initial_state contains the last_committed_seq after WAL replay. in no-wal mode, we can simply fallback - // to last_l0_seq. let initial_durable_seq = initial_state .last_remote_persisted_seq .max(initial_state.core().last_l0_seq); - let status_manager = DbStatusManager::new_with_initial_values( - initial_durable_seq, + status_manager.report_durable_seq( + initial_state + .last_remote_persisted_seq + .max(initial_state.core().last_l0_seq), + ); + status_manager.report_manifest_and_memtable_segments( VersionedManifest::from(initial_state.as_ref()), collect_touched_segments(initial_state.as_ref()), ); @@ -479,6 +480,7 @@ impl DbReaderInner { let inner = Self { manifest_store, table_store, + wal_reader, options, mode, state, @@ -489,7 +491,6 @@ impl DbReaderInner { status_manager, segment_extractor, rand, - replay_tasks, recorder, }; Ok(inner) @@ -704,22 +705,58 @@ impl DbReaderInner { .report_manifest_and_memtable_segments(versioned_manifest, touched_segments); } + async fn maybe_replay_new_wals(&self) -> Result<(), SlateDBError> { + if self.options.skip_wal_replay { + return Ok(()); + } + let current_state = Arc::clone(&self.state.read()); + let mut imm_memtable = current_state.imm_memtable.clone(); + let generation = Arc::clone(¤t_state.generation); + let mut publish = + |imm_memtable: &ReplayMemtables, last_wal_id: u64, last_committed_seq: u64| { + self.oracle.advance_durable_seq(last_committed_seq); + self.db_stats + .reader_replay_memtables + .set(imm_memtable.len() as i64); + let mut write_guard = self.state.write(); + *write_guard = Arc::new(ReaderState { + generation: Arc::clone(&generation), + imm_memtable: imm_memtable.clone(), + last_wal_id, + last_remote_persisted_seq: last_committed_seq, + }); + drop(write_guard); + self.status_manager + .report_memtable_segments(collect_touched_segments(self.state.read().as_ref())); + }; + + Self::replay_wal_into( + Arc::clone(&self.table_store), + self.wal_reader.as_ref(), + &self.options, + current_state.core(), + &mut imm_memtable, + Some(( + current_state.last_wal_id, + current_state.last_remote_persisted_seq, + )), + WalReplayEnd::Latest, + self.segment_extractor.as_ref(), + Some(&mut publish), + Some(&self.db_stats), + ) + .await?; + Ok(()) + } + async fn rebuild_checkpoint_state( &self, new_checkpoint: Checkpoint, ) -> Result { let manifest_id = new_checkpoint.manifest_id; let manifest = self.manifest_store.read_manifest(manifest_id).await?; - self.rebuild_state( - Some(new_checkpoint), - manifest_id, - manifest, - ReaderReplayPath::ExactOpen { - include_unmanifested: !self.options.skip_wal_replay, - source: ExactWalReplaySource::CheckpointRecovery, - }, - ) - .await + self.rebuild_state(Some(new_checkpoint), manifest_id, manifest) + .await } async fn rebuild_state( @@ -727,29 +764,8 @@ impl DbReaderInner { checkpoint: Option, manifest_id: u64, manifest: Manifest, - replay_path: ReaderReplayPath, ) -> Result { let prior = self.state.read().clone(); - self.rebuild_state_from( - prior, - checkpoint, - manifest_id, - manifest, - replay_path, - self.replay_tasks.clone(), - ) - .await - } - - async fn rebuild_state_from( - &self, - prior: Arc, - checkpoint: Option, - manifest_id: u64, - manifest: Manifest, - replay_path: ReaderReplayPath, - replay_tasks: ReplayTaskScope, - ) -> Result { let replay_cursor = Some(( prior.last_wal_id.max(manifest.core.replay_after_wal_id), prior @@ -788,95 +804,49 @@ impl DbReaderInner { manifest_id, manifest, imm_memtable, - replay_path, + WalReplayEnd::for_reader(self.mode, &self.options), Arc::clone(&self.table_store), + self.wal_reader.as_ref(), &self.options, self.segment_extractor.as_ref(), replay_cursor, &self.db_stats, - Some(replay_tasks), - &self.system_clock, ) .await } - async fn build_latest_manifest_candidate( - &self, - base: Arc, - replay_tasks: ReplayTaskScope, - ) -> Result, SlateDBError> { - let latest_manifest = self.manifest_store.read_latest_manifest().await?; - if latest_manifest.id <= base.generation.manifest_id { - return Ok(None); - } - - let base_manifest_id = base.generation.manifest_id; - let base_last_wal_id = base.last_wal_id; - let state = self - .rebuild_state_from( - base, - None, - latest_manifest.id, - latest_manifest.manifest, - ReaderReplayPath::RuntimeManifest, - replay_tasks, - ) - .await?; - Ok(Some(ManifestRefreshCandidate { - base_manifest_id, - base_last_wal_id, - state, - })) - } - async fn build_reader_state( checkpoint: Option, manifest_id: u64, manifest: Manifest, mut imm_memtable: ReplayMemtables, - replay_path: ReaderReplayPath, + replay_wals: Option, table_store: Arc, + wal_reader: &dyn WalReaderTrait, options: &DbReaderOptions, segment_extractor: Option<&Arc>, replay_cursor: Option<(u64, u64)>, db_stats: &DbStats, - replay_tasks: Option, - system_clock: &Arc, ) -> Result { - let (last_wal_id, last_committed_seq) = match replay_path { - ReaderReplayPath::ExactOpen { - include_unmanifested, - source, - } => { - Self::replay_wal_into_exact_with_source( + let (last_wal_id, last_committed_seq) = match replay_wals { + Some(replay_end) => { + Self::replay_wal_into( Arc::clone(&table_store), + wal_reader, options, &manifest.core, &mut imm_memtable, replay_cursor, - include_unmanifested, - source, + replay_end, segment_extractor, None, Some(db_stats), - replay_tasks.clone(), - ) - .await? - } - ReaderReplayPath::RuntimeManifest => { - Self::replay_manifest_range_into( - Arc::clone(&table_store), - options, - &manifest.core, - &mut imm_memtable, - replay_cursor, - segment_extractor, - Some(db_stats), - replay_tasks, - system_clock, ) .await? } + // Skipping replay reads no WAL at all: the reader stays at the + // watermark it has already reached (the most recently read manifest) + None => Self::replayed_watermark(&manifest.core, &imm_memtable), }; db_stats @@ -891,23 +861,22 @@ impl DbReaderInner { }) } - #[cfg(test)] + async fn refresh_latest_manifest(&self) -> Result<(), SlateDBError> { + let latest_manifest = self.manifest_store.read_latest_manifest().await?; + self.apply_latest_manifest(latest_manifest).await + } + async fn apply_latest_manifest( &self, latest_manifest: VersionedManifest, ) -> Result<(), SlateDBError> { let manifest_id = latest_manifest.id; if manifest_id <= self.state.read().generation.manifest_id { - return Ok(()); + return self.maybe_replay_new_wals().await; } let new_state = self - .rebuild_state( - None, - manifest_id, - latest_manifest.manifest, - ReaderReplayPath::RuntimeManifest, - ) + .rebuild_state(None, manifest_id, latest_manifest.manifest) .await?; self.install_state(new_state); info!("refreshed reader to latest manifest [manifest_id={manifest_id}]"); @@ -974,8 +943,7 @@ impl DbReaderInner { self: &Arc, task_executor: &MessageHandlerExecutor, ) -> Result<(), SlateDBError> { - let (tail_tx, tail_rx) = async_channel::unbounded(); - let poller = ManifestPoller::new(Arc::clone(self), tail_tx, tail_rx); + let poller = ManifestPoller::new(Arc::clone(self)); let (_tx, rx) = async_channel::unbounded(); let result = task_executor.add_handler( DB_READER_TASK_NAME.to_string(), @@ -987,104 +955,75 @@ impl DbReaderInner { result } - #[cfg(test)] - async fn replay_wal_into_exact( - table_store: Arc, - reader_options: &DbReaderOptions, - core: &ManifestCore, - into_tables: &mut ReplayMemtables, - replay_cursor: Option<(u64, u64)>, - replay_new_wals: bool, - segment_extractor: Option<&Arc>, - publish: Option>, - db_stats: Option<&DbStats>, - ) -> Result<(u64, u64), SlateDBError> { - Self::replay_wal_into_exact_with_source( - table_store, - reader_options, - core, - into_tables, - replay_cursor, - replay_new_wals, - ExactWalReplaySource::ReaderOpen, - segment_extractor, - publish, - db_stats, - None, - ) - .await + /// The `(last replayed WAL id, last committed seq)` the reader has already + /// reached: the watermark of the most recently replayed table, or the + /// manifest's own boundary when nothing has been replayed into `tables`. + fn replayed_watermark(core: &ManifestCore, tables: &ReplayMemtables) -> (u64, u64) { + match tables.front() { + Some(latest_replayed_table) => ( + latest_replayed_table.recent_flushed_wal_id(), + latest_replayed_table.table().last_seq().unwrap_or(0), + ), + None => (core.replay_after_wal_id, core.last_l0_seq), + } } - async fn replay_wal_into_exact_with_source( + async fn replay_wal_into( table_store: Arc, + wal_reader: &dyn WalReaderTrait, reader_options: &DbReaderOptions, core: &ManifestCore, into_tables: &mut ReplayMemtables, replay_cursor: Option<(u64, u64)>, - replay_new_wals: bool, - source: ExactWalReplaySource, + replay_end: WalReplayEnd, segment_extractor: Option<&Arc>, mut publish: Option>, db_stats: Option<&DbStats>, - replay_tasks: Option, ) -> Result<(u64, u64), SlateDBError> { let (mut replay_after_wal_id, mut last_committed_seq) = - replay_cursor.unwrap_or_else(|| { - if let Some(latest_replayed_table) = into_tables.front() { - ( - latest_replayed_table.recent_flushed_wal_id(), - latest_replayed_table.table().last_seq().unwrap_or(0), - ) - } else { - (core.replay_after_wal_id, core.last_l0_seq) - } - }); - let replay_list_counter = db_stats.map(|stats| match source { - ExactWalReplaySource::ReaderOpen => { - Arc::clone(&stats.reader_wal_replay_list_reader_open) - } - ExactWalReplaySource::CheckpointRecovery => { - Arc::clone(&stats.reader_wal_replay_list_checkpoint_recovery) - } - ExactWalReplaySource::WriterOpen => { - unreachable!("writer replay does not use DbReader metrics") - } - }); - let wal_id_end = if replay_new_wals { - if let Some(counter) = replay_list_counter.as_ref() { - counter.increment(1); - } - table_store.last_seen_wal_id(replay_after_wal_id).await? + 1 - } else { - core.next_wal_sst_id + replay_cursor.unwrap_or_else(|| Self::replayed_watermark(core, into_tables)); + let wal_id_start = replay_after_wal_id + .checked_add(1) + .ok_or(SlateDBError::InvalidDBState)?; + let wal_id_end = match replay_end { + WalReplayEnd::Manifest => core.next_wal_sst_id, + WalReplayEnd::Latest => wal_reader + .last_wal_file_id(replay_after_wal_id) + .await? + .checked_add(1) + .ok_or(SlateDBError::InvalidDBState)?, }; + if wal_id_start >= wal_id_end { + return Ok((replay_after_wal_id, last_committed_seq)); + } let replay_options = WalReplayOptions { - prefetch: reader_options.wal_replay, max_memtable_bytes: reader_options.max_memtable_bytes as usize, - // Skip entries that we already have in `imm_memtable` (that might be above last_l0_seq). + // Skip entries that we already have in `imm_memtable` (that might be above + // last_l0_seq). min_seq: Some(last_committed_seq), - source, - task_scope: replay_tasks, }; - if let Some(counter) = replay_list_counter { - counter.increment(1); - } - let mut replay_iter = ExactWalReplayIterator::range( - (replay_after_wal_id + 1)..wal_id_end, + let wal_iter = wal_reader + .iterator((wal_id_start..wal_id_end).into()) + .await?; + let mut replay_iter = WalReplayIterator::for_wal_iterator( + wal_iter, core, replay_options, Arc::clone(&table_store), - ) - .await?; + )?; while let Some(replayed_table) = match replay_iter.next().await { Ok(Some(replayed_table)) => Some(replayed_table), Ok(None) => None, + Err(SlateDBError::WalTruncated(_)) => None, Err(err) => return Err(err), } { - assert!(replayed_table.last_wal_id > replay_after_wal_id); + // `last_wal_id` is a conservative watermark: a table that ends mid-file + // is tagged with the last fully replayed WAL ID, which may equal the + // watermark of the previous table. + assert!(replayed_table.last_wal_id >= replay_after_wal_id); let replayed_ssts = replayed_table.last_wal_id - replay_after_wal_id; replay_after_wal_id = replayed_table.last_wal_id; if let Some(db_stats) = db_stats { @@ -1123,277 +1062,6 @@ impl DbReaderInner { Ok((replay_after_wal_id, last_committed_seq)) } - async fn replay_manifest_range_into( - table_store: Arc, - reader_options: &DbReaderOptions, - core: &ManifestCore, - into_tables: &mut ReplayMemtables, - replay_cursor: Option<(u64, u64)>, - segment_extractor: Option<&Arc>, - db_stats: Option<&DbStats>, - replay_tasks: Option, - system_clock: &Arc, - ) -> Result<(u64, u64), SlateDBError> { - let mut replay_cursor = - replay_cursor.unwrap_or((core.replay_after_wal_id, core.last_l0_seq)); - while replay_cursor.0.saturating_add(1) < core.next_wal_sst_id { - let start = replay_cursor.0.saturating_add(1); - let deadline = system_clock.now() + MAX_RUNTIME_REPLAY_TURN_TIME; - let completed_before = replay_cursor.0; - let mut replay = RuntimeWalReplayIterator::range( - start..core.next_wal_sst_id, - core, - RuntimeWalReplayOptions { - max_memtable_bytes: reader_options.max_memtable_bytes as usize, - min_seq: Some(replay_cursor.1), - max_wals_per_batch: std::num::NonZeroUsize::new( - usize::try_from(MAX_RUNTIME_WALS_PER_TURN) - .expect("runtime WAL turn limit must fit usize"), - ), - source: RuntimeWalReplaySource::Manifest, - task_scope: replay_tasks.clone(), - ..RuntimeWalReplayOptions::default() - }, - Arc::clone(&table_store), - )?; - - loop { - let completed_wals = replay_cursor.0.saturating_sub(completed_before); - if completed_wals >= MAX_RUNTIME_WALS_PER_TURN { - break; - } - let remaining = deadline - .signed_duration_since(system_clock.now()) - .to_std() - .unwrap_or(std::time::Duration::ZERO); - let replayed = tokio::select! { - biased; - _ = system_clock.sleep(remaining) => break, - result = replay.next() => result, - }; - if system_clock.now() >= deadline { - break; - } - let replayed_table = match replayed { - Ok(Some(table)) => table, - Ok(None) => break, - Err(RuntimeWalReplayError::MissingInitialObject { source, .. }) - | Err(RuntimeWalReplayError::Replay(source)) => return Err(source), - }; - Self::apply_runtime_replayed_table( - replayed_table, - into_tables, - &mut replay_cursor, - segment_extractor, - db_stats, - )?; - } - drop(replay); - - if replay_cursor.0.saturating_add(1) < core.next_wal_sst_id { - tokio::task::yield_now().await; - } - } - Ok(replay_cursor) - } - - #[cfg(test)] - async fn replay_runtime_range_into( - table_store: Arc, - reader_options: &DbReaderOptions, - core: &ManifestCore, - into_tables: &mut ReplayMemtables, - replay_cursor: &mut (u64, u64), - wal_range: std::ops::Range, - missing_policy: RuntimeMissingPolicy, - segment_extractor: Option<&Arc>, - db_stats: Option<&DbStats>, - max_wals_per_batch: Option, - replay_tasks: Option, - ) -> Result { - let mut replay = RuntimeWalReplayIterator::range( - wal_range, - core, - RuntimeWalReplayOptions { - max_memtable_bytes: reader_options.max_memtable_bytes as usize, - min_seq: Some(replay_cursor.1), - max_wals_per_batch, - source: max_wals_per_batch.map_or(RuntimeWalReplaySource::Manifest, |_| { - RuntimeWalReplaySource::Tail - }), - task_scope: replay_tasks, - ..RuntimeWalReplayOptions::default() - }, - table_store, - )?; - - loop { - let replayed_table = match replay.next().await { - Ok(Some(table)) => table, - Ok(None) => break, - Err(RuntimeWalReplayError::MissingInitialObject { wal_id, source }) => { - if missing_policy == RuntimeMissingPolicy::ExactNextIsCaughtUp(wal_id) { - return Ok(true); - } - return Err(source); - } - Err(RuntimeWalReplayError::Replay(source)) => return Err(source), - }; - Self::apply_runtime_replayed_table( - replayed_table, - into_tables, - replay_cursor, - segment_extractor, - db_stats, - )?; - } - - Ok(false) - } - - fn apply_runtime_replayed_table( - replayed_table: ReplayedMemtable, - into_tables: &mut ReplayMemtables, - replay_cursor: &mut (u64, u64), - segment_extractor: Option<&Arc>, - db_stats: Option<&DbStats>, - ) -> Result { - assert!(replayed_table.last_wal_id > replay_cursor.0); - let replayed_ssts = replayed_table.last_wal_id - replay_cursor.0; - replay_cursor.0 = replayed_table.last_wal_id; - if let Some(db_stats) = db_stats { - let metadata = replayed_table.table.metadata(); - db_stats.reader_wal_replay_ssts.increment(replayed_ssts); - db_stats - .reader_wal_replay_bytes - .increment(metadata.entries_size_in_bytes as u64); - db_stats.reader_wal_replay_batches.increment(1); - } - if !replayed_table.table.is_empty() && replayed_table.last_seq > replay_cursor.1 { - let first_seq = replayed_table - .table - .table() - .first_seq() - .expect("expected first_seq on non-empty table"); - assert!(first_seq > replay_cursor.1); - replay_cursor.1 = replayed_table.last_seq; - if let Some(extractor) = segment_extractor { - Self::record_replayed_touched_segments(extractor.as_ref(), &replayed_table.table)?; - } - into_tables.prepend(Arc::new(ImmutableMemtable::new( - replayed_table.table, - replayed_table.last_wal_id, - ))); - } - Ok(replayed_ssts) - } - - async fn replay_runtime_turn( - inner: Arc, - base: Arc, - replay_tasks: ReplayTaskScope, - ) -> Result { - let manifest_id = base.generation.manifest_id; - let base_last_wal_id = base.last_wal_id; - let core = base.core(); - let known_end = core.next_wal_sst_id; - let deadline = inner.system_clock.now() + MAX_RUNTIME_REPLAY_TURN_TIME; - let mut imm_memtable = base.imm_memtable.clone(); - let mut cursor = (base.last_wal_id, base.last_remote_persisted_seq); - let mut completed_wals = 0_u64; - let mut caught_up = false; - - 'turn: while completed_wals < MAX_RUNTIME_WALS_PER_TURN - && inner.system_clock.now() < deadline - { - let next_wal_id = cursor - .0 - .checked_add(1) - .ok_or(SlateDBError::InvalidDBState)?; - let remaining = MAX_RUNTIME_WALS_PER_TURN - completed_wals; - let (range_end, missing_policy) = if next_wal_id < known_end { - ( - next_wal_id - .checked_add(remaining) - .ok_or(SlateDBError::InvalidDBState)? - .min(known_end), - RuntimeMissingPolicy::Error, - ) - } else { - ( - next_wal_id - .checked_add(1) - .ok_or(SlateDBError::InvalidDBState)?, - RuntimeMissingPolicy::ExactNextIsCaughtUp(next_wal_id), - ) - }; - let mut replay = RuntimeWalReplayIterator::range( - next_wal_id..range_end, - core, - RuntimeWalReplayOptions { - max_memtable_bytes: inner.options.max_memtable_bytes as usize, - min_seq: Some(cursor.1), - max_wals_per_batch: std::num::NonZeroUsize::new(1), - source: RuntimeWalReplaySource::Tail, - task_scope: Some(replay_tasks.clone()), - ..RuntimeWalReplayOptions::default() - }, - Arc::clone(&inner.table_store), - )?; - - loop { - let remaining = deadline - .signed_duration_since(inner.system_clock.now()) - .to_std() - .unwrap_or(std::time::Duration::ZERO); - let replayed = tokio::select! { - biased; - _ = inner.system_clock.sleep(remaining) => break 'turn, - result = replay.next() => result, - }; - // A result completing at the exact deadline is deliberately - // discarded even if the runtime happened to poll it first. - if inner.system_clock.now() >= deadline { - break 'turn; - } - let replayed_table = match replayed { - Ok(Some(table)) => table, - Ok(None) => break, - Err(RuntimeWalReplayError::MissingInitialObject { wal_id, source }) => { - if missing_policy == RuntimeMissingPolicy::ExactNextIsCaughtUp(wal_id) { - caught_up = true; - break 'turn; - } - return Err(source); - } - Err(RuntimeWalReplayError::Replay(source)) => return Err(source), - }; - completed_wals = completed_wals.saturating_add(Self::apply_runtime_replayed_table( - replayed_table, - &mut imm_memtable, - &mut cursor, - inner.segment_extractor.as_ref(), - Some(&inner.db_stats), - )?); - if completed_wals >= MAX_RUNTIME_WALS_PER_TURN { - break 'turn; - } - } - } - - Ok(RuntimeReplayTurn { - manifest_id, - base_last_wal_id, - imm_memtable, - last_wal_id: cursor.0, - last_committed_seq: cursor.1, - caught_up, - reached_turn_limit: !caught_up - && (completed_wals >= MAX_RUNTIME_WALS_PER_TURN - || inner.system_clock.now() >= deadline), - }) - } - /// Re-derive each replayed entry's segment prefix (RFC-0024) and record the /// table's touched-segment set, mirroring the writer's replay path. Durable /// WAL entries were validated when accepted, so the antichain check is not @@ -1424,8 +1092,8 @@ impl DbReaderInner { /// /// ## Returns /// - `Ok(())` if the reader is still open. - /// - `Err(SlateDBError::Closed)` if the reader was closed successfully - /// (state.result_reader() returns Ok(())). + /// - `Err(SlateDBError::Closed)` if the reader was closed successfully (state.result_reader() + /// returns Ok(())). /// - `Err(e)` if the reader was closed with an error, where `e` is the error /// (state.result_reader() returns Err(e)). pub(crate) fn check_closed(&self) -> Result<(), SlateDBError> { @@ -1443,23 +1111,10 @@ impl DbReaderInner { struct ManifestPoller { inner: Arc, generations: HashMap>, - tail_tx: async_channel::Sender, - tail_rx: Option>, - tail_in_flight: bool, - continuation_queued: bool, - tail_task: Option>, - tail_scope: Option, - manifest_refresh_in_flight: bool, - manifest_refresh_task: Option>, - manifest_refresh_scope: Option, } impl ManifestPoller { - fn new( - inner: Arc, - tail_tx: async_channel::Sender, - tail_rx: async_channel::Receiver, - ) -> Self { + fn new(inner: Arc) -> Self { let mut generations = HashMap::new(); if inner.mode == DbReaderMode::ManagedCheckpoint { let generation = Arc::clone(&inner.state.read().generation); @@ -1469,244 +1124,44 @@ impl ManifestPoller { .id; generations.insert(checkpoint_id, Arc::downgrade(&generation)); } - let poller = Self { - inner, - generations, - tail_rx: Some(tail_rx), - tail_tx, - tail_in_flight: false, - continuation_queued: false, - tail_task: None, - tail_scope: None, - manifest_refresh_in_flight: false, - manifest_refresh_task: None, - manifest_refresh_scope: None, - }; + let poller = Self { inner, generations }; poller.report_active_checkpoints(); poller } - fn runtime_tailing_enabled(&self) -> bool { - !self.inner.options.skip_wal_replay - && !matches!(self.inner.mode, DbReaderMode::Checkpoint(_)) + fn report_active_checkpoints(&self) { + self.inner + .db_stats + .reader_active_checkpoints + .set(self.generations.len() as i64); } - fn start_runtime_tail(&mut self) { - if self.tail_in_flight || self.manifest_refresh_in_flight || !self.runtime_tailing_enabled() - { - return; - } - self.tail_in_flight = true; - self.continuation_queued = false; - let inner = Arc::clone(&self.inner); - let base = Arc::clone(&inner.state.read()); - let sender = self.tail_tx.clone(); - let scope = inner.replay_tasks.child(); - let replay_scope = scope.clone(); - self.tail_task = Some(scope.spawn(async move { - let replay = tokio::select! { - biased; - _ = replay_scope.cancelled() => return, - replay = AssertUnwindSafe(DbReaderInner::replay_runtime_turn( - inner, - base, - replay_scope.clone(), - )).catch_unwind() => replay, - }; - let result = match replay { - Ok(result) => result, - Err(panic) => { - let task_name = "reader_runtime_wal_replay".to_string(); - error!( - "runtime WAL replay task panicked [task_name={}, panic={}]", - task_name, - panic_string(&panic), - ); - Err(SlateDBError::BackgroundTaskPanic(task_name)) - } - }; - let _ = sender.send(DbReaderMessage::WalTailComplete(result)).await; - })); - self.tail_scope = Some(scope); + fn register_current_generation(&mut self) { + let generation = Arc::clone(&self.inner.state.read().generation); + let checkpoint_id = generation + .checkpoint() + .expect("managed reader must have a checkpoint") + .id; + self.generations + .insert(checkpoint_id, Arc::downgrade(&generation)); + self.report_active_checkpoints(); } - fn start_manifest_refresh(&mut self) { - if self.manifest_refresh_in_flight { - return; + async fn delete_released_checkpoints( + &mut self, + manifest: &mut StoredManifest, + ) -> Result<(), SlateDBError> { + let released = self + .generations + .iter() + .filter_map(|(id, generation)| generation.upgrade().is_none().then_some(*id)) + .collect::>(); + if released.is_empty() { + return Ok(()); } - self.manifest_refresh_in_flight = true; - let inner = Arc::clone(&self.inner); - let base = Arc::clone(&inner.state.read()); - let sender = self.tail_tx.clone(); - let scope = inner.replay_tasks.child(); - let replay_scope = scope.clone(); - self.manifest_refresh_task = Some(scope.spawn(async move { - let result = tokio::select! { - biased; - _ = replay_scope.cancelled() => return, - result = AssertUnwindSafe(inner.build_latest_manifest_candidate( - base, - replay_scope.clone(), - )).catch_unwind() => match result { - Ok(result) => result, - Err(panic) => { - let task_name = "reader_manifest_wal_replay".to_string(); - error!( - "manifest WAL replay task panicked [task_name={}, panic={}]", - task_name, - panic_string(&panic), - ); - Err(SlateDBError::BackgroundTaskPanic(task_name)) - } - }, - }; - let _ = sender - .send(DbReaderMessage::ManifestRefreshComplete(result)) - .await; - })); - self.manifest_refresh_scope = Some(scope); - } - - fn schedule_tail_continuation(&mut self) { - if self.continuation_queued || !self.runtime_tailing_enabled() { - return; - } - self.continuation_queued = true; - let _ = self.tail_tx.try_send(DbReaderMessage::ContinueWalTail); - } - - async fn finish_tail_task(&mut self) { - if let Some(task) = self.tail_task.take() { - let _ = task.await; - } - if let Some(scope) = self.tail_scope.take() { - scope.shutdown().await; - } - } - - async fn cancel_tail_task(&mut self) { - if let Some(scope) = self.tail_scope.as_ref() { - scope.cancel(); - } - if let Some(task) = self.tail_task.as_ref() { - task.abort(); - } - self.finish_tail_task().await; - self.tail_in_flight = false; - } - - async fn finish_manifest_refresh_task(&mut self) { - if let Some(task) = self.manifest_refresh_task.take() { - let _ = task.await; - } - if let Some(scope) = self.manifest_refresh_scope.take() { - scope.shutdown().await; - } - } - - async fn cancel_manifest_refresh_task(&mut self) { - if let Some(scope) = self.manifest_refresh_scope.as_ref() { - scope.cancel(); - } - if let Some(task) = self.manifest_refresh_task.as_ref() { - task.abort(); - } - self.finish_manifest_refresh_task().await; - self.manifest_refresh_in_flight = false; - } - - async fn apply_manifest_refresh(&mut self, candidate: Option) { - self.manifest_refresh_in_flight = false; - self.finish_manifest_refresh_task().await; - let Some(candidate) = candidate else { - self.inner.db_stats.reader_manifest_polls.increment(1); - self.schedule_tail_continuation(); - return; - }; - let current = Arc::clone(&self.inner.state.read()); - if current.generation.manifest_id != candidate.base_manifest_id - || current.last_wal_id != candidate.base_last_wal_id - { - self.start_manifest_refresh(); - return; - } - - let manifest_id = candidate.state.generation.manifest_id; - self.inner.install_state(candidate.state); - self.inner.db_stats.reader_manifest_polls.increment(1); - info!("refreshed reader to latest manifest [manifest_id={manifest_id}]"); - self.schedule_tail_continuation(); - } - - async fn apply_runtime_tail(&mut self, turn: RuntimeReplayTurn) { - self.tail_in_flight = false; - self.finish_tail_task().await; - let current = Arc::clone(&self.inner.state.read()); - if current.generation.manifest_id != turn.manifest_id - || current.last_wal_id != turn.base_last_wal_id - { - self.schedule_tail_continuation(); - return; - } - - let generation = Arc::clone(¤t.generation); - self.inner - .oracle - .advance_durable_seq(turn.last_committed_seq); - self.inner - .db_stats - .reader_replay_memtables - .set(turn.imm_memtable.len() as i64); - let new_state = Arc::new(ReaderState { - generation, - imm_memtable: turn.imm_memtable, - last_wal_id: turn.last_wal_id, - last_remote_persisted_seq: turn.last_committed_seq, - }); - let touched_segments = collect_touched_segments(new_state.as_ref()); - *self.inner.state.write() = new_state; - self.inner - .status_manager - .report_memtable_segments(touched_segments); - - if turn.reached_turn_limit || !turn.caught_up { - self.schedule_tail_continuation(); - } - } - - fn report_active_checkpoints(&self) { - self.inner - .db_stats - .reader_active_checkpoints - .set(self.generations.len() as i64); - } - - fn register_current_generation(&mut self) { - let generation = Arc::clone(&self.inner.state.read().generation); - let checkpoint_id = generation - .checkpoint() - .expect("managed reader must have a checkpoint") - .id; - self.generations - .insert(checkpoint_id, Arc::downgrade(&generation)); - self.report_active_checkpoints(); - } - - async fn delete_released_checkpoints( - &mut self, - manifest: &mut StoredManifest, - ) -> Result<(), SlateDBError> { - let released = self - .generations - .iter() - .filter_map(|(id, generation)| generation.upgrade().is_none().then_some(*id)) - .collect::>(); - if released.is_empty() { - return Ok(()); - } - manifest.delete_checkpoints(&released).await?; - for id in released { - self.generations.remove(&id); + manifest.delete_checkpoints(&released).await?; + for id in released { + self.generations.remove(&id); } self.report_active_checkpoints(); Ok(()) @@ -1789,34 +1244,8 @@ impl ManifestPoller { } } -struct TailReplayNotifier { - receiver: async_channel::Receiver, -} - -#[async_trait] -impl Notifier for TailReplayNotifier { - async fn notify(&mut self) -> DbReaderMessage { - match self.receiver.recv().await { - Ok(message) => message, - Err(_) => std::future::pending().await, - } - } -} - impl Drop for ManifestPoller { fn drop(&mut self) { - if let Some(scope) = self.tail_scope.take() { - scope.cancel(); - } - if let Some(task) = self.tail_task.take() { - task.abort(); - } - if let Some(scope) = self.manifest_refresh_scope.take() { - scope.cancel(); - } - if let Some(task) = self.manifest_refresh_task.take() { - task.abort(); - } // The gauge tracks only GC checkpoints actively managed by this // poller. Reset it even if startup, cleanup, or the poller task fails. self.inner.db_stats.reader_active_checkpoints.set(0); @@ -1826,61 +1255,14 @@ impl Drop for ManifestPoller { #[async_trait] impl MessageHandler for ManifestPoller { fn tickers(&mut self) -> Vec> { - let mut tickers = vec![MessageTickerDef::new( + vec![MessageTickerDef::new( self.inner.options.manifest_poll_interval, Box::new(|| DbReaderMessage::PollManifest), - )]; - if self.runtime_tailing_enabled() { - tickers.push(MessageTickerDef::new( - self.inner.options.wal_poll_interval, - Box::new(|| DbReaderMessage::PollWalTail), - )); - } - tickers - } - - fn notifiers(&mut self) -> Vec>> { - self.tail_rx.take().map_or_else(Vec::new, |receiver| { - vec![Box::new(TailReplayNotifier { receiver }) as Box>] - }) + )] } async fn handle(&mut self, message: DbReaderMessage) -> Result<(), SlateDBError> { - let replay_tasks = self.inner.replay_tasks.clone(); - tokio::select! { - biased; - _ = replay_tasks.cancelled() => { - self.cancel_tail_task().await; - self.cancel_manifest_refresh_task().await; - Ok(()) - } - result = async { - match message { - DbReaderMessage::PollWalTail | DbReaderMessage::ContinueWalTail => { - self.continuation_queued = false; - self.start_runtime_tail(); - return Ok(()); - } - DbReaderMessage::WalTailComplete(result) => { - let turn = result?; - self.apply_runtime_tail(turn).await; - return Ok(()); - } - DbReaderMessage::ManifestRefreshComplete(result) => { - match result { - Ok(candidate) => self.apply_manifest_refresh(candidate).await, - Err(error) => { - self.manifest_refresh_in_flight = false; - self.finish_manifest_refresh_task().await; - warn!( - "failed to refresh reader to latest manifest [error={error:?}]" - ); - } - } - return Ok(()); - } - DbReaderMessage::PollManifest => {} - } + assert!(matches!(message, DbReaderMessage::PollManifest)); match self.inner.mode { DbReaderMode::ManagedCheckpoint => { let mut manifest = StoredManifest::load( @@ -1898,7 +1280,8 @@ impl MessageHandler for ManifestPoller { let checkpoint = self.inner.create_checkpoint(&mut manifest).await?; self.inner.reestablish_checkpoint(checkpoint).await?; self.register_current_generation(); - self.schedule_tail_continuation(); + } else { + self.inner.maybe_replay_new_wals().await?; } self.refresh_live_checkpoints(&mut manifest).await?; @@ -1906,15 +1289,17 @@ impl MessageHandler for ManifestPoller { Ok(()) } DbReaderMode::FollowLatest => { - self.cancel_tail_task().await; - self.start_manifest_refresh(); + let result = self.inner.refresh_latest_manifest().await; + if let Err(error) = result { + warn!("failed to refresh reader to latest manifest [error={error:?}]"); + } else { + self.inner.db_stats.reader_manifest_polls.increment(1); + } Ok(()) } // No polling is needed for a pinned checkpoint, so we just return Ok(()). DbReaderMode::Checkpoint(_) => Ok(()), } - } => result, - } } async fn cleanup( @@ -1922,9 +1307,16 @@ impl MessageHandler for ManifestPoller { _messages: BoxStream<'async_trait, DbReaderMessage>, _result: Result<(), SlateDBError>, ) -> Result<(), SlateDBError> { - self.cancel_tail_task().await; - self.cancel_manifest_refresh_task().await; - if self.inner.mode == DbReaderMode::ManagedCheckpoint { + if self.inner.mode != DbReaderMode::ManagedCheckpoint { + return Ok(()); + } + let mut manifest = StoredManifest::load( + Arc::clone(&self.inner.manifest_store), + self.inner.system_clock.clone(), + ) + .await?; + let checkpoint_ids = self.generations.keys().copied().collect::>(); + if !checkpoint_ids.is_empty() { let live_generations = self .generations .values() @@ -1939,10 +1331,11 @@ impl MessageHandler for ManifestPoller { for generation in live_generations { generation.drain().await; } - // Managed checkpoints have a finite lease and can safely expire. - // Remote cleanup is intentionally not part of reader close: an - // unavailable object store must not prevent retirement proof. - self.generations.clear(); + info!( + "deleting reader established checkpoints for shutdown [checkpoint_ids={:?}]", + checkpoint_ids + ); + manifest.delete_checkpoints(&checkpoint_ids).await?; } self.inner.db_stats.reader_active_checkpoints.set(0); Ok(()) @@ -1951,12 +1344,6 @@ impl MessageHandler for ManifestPoller { impl DbReader { fn validate_options(mode: DbReaderMode, options: &DbReaderOptions) -> Result<(), SlateDBError> { - options.wal_replay.validate()?; - if options.wal_poll_interval.is_zero() { - return Err(SlateDBError::InvalidWalPollInterval( - options.wal_poll_interval, - )); - } if mode != DbReaderMode::ManagedCheckpoint { return Ok(()); } @@ -1982,7 +1369,7 @@ impl DbReader { pub(crate) async fn preload_cache( &self, cached_obj_store: &CachedObjectStore, - path: object_store::path::Path, + path: Path, ) -> Result<(), SlateDBError> { let state = Arc::clone(&self.inner.state.read()); let external_ssts = state.generation.manifest.external_ssts(); @@ -2069,9 +1456,13 @@ impl DbReader { /// # Examples /// /// ``` - /// use slatedb::{Db, DbReader, Error}; - /// use slatedb::object_store::{ObjectStore, memory::InMemory}; - /// use std::sync::Arc; + /// use { + /// slatedb::{ + /// object_store::{memory::InMemory, ObjectStore}, + /// Db, DbReader, Error, + /// }, + /// std::sync::Arc, + /// }; /// /// #[tokio::main] /// async fn main() -> Result<(), Error> { @@ -2080,9 +1471,7 @@ impl DbReader { /// let db = Db::open("test_db", Arc::clone(&object_store)).await?; /// db.close().await?; /// // Then open a reader - /// let reader = DbReader::builder("test_db", object_store) - /// .build() - /// .await?; + /// let reader = DbReader::builder("test_db", object_store).build().await?; /// Ok(()) /// } /// ``` @@ -2096,7 +1485,9 @@ impl DbReader { pub(crate) async fn open_internal( manifest_store: Arc, table_store: Arc, + wal_store: Arc, mode: DbReaderMode, + wal_reader: Option>, merge_operator: Option, segment_extractor: Option>, options: DbReaderOptions, @@ -2126,6 +1517,8 @@ impl DbReader { DbReaderInner::new( manifest_store, table_store, + wal_store, + wal_reader, options, mode, merge_operator, @@ -2200,9 +1593,14 @@ impl DbReader { /// ## Examples /// /// ``` - /// use slatedb::{Db, DbReader, DbReaderMode, config::DbReaderOptions, Error}; - /// use slatedb::object_store::{ObjectStore, memory::InMemory}; - /// use std::sync::Arc; + /// use { + /// slatedb::{ + /// config::DbReaderOptions, + /// object_store::{memory::InMemory, ObjectStore}, + /// Db, DbReader, DbReaderMode, Error, + /// }, + /// std::sync::Arc, + /// }; /// /// #[tokio::main] /// async fn main() -> Result<(), Error> { @@ -2212,11 +1610,12 @@ impl DbReader { /// db.flush().await?; /// /// let reader = DbReader::open( - /// "test_db", - /// Arc::clone(&object_store), - /// DbReaderMode::ManagedCheckpoint, - /// DbReaderOptions::default(), - /// ).await?; + /// "test_db", + /// Arc::clone(&object_store), + /// DbReaderMode::ManagedCheckpoint, + /// DbReaderOptions::default(), + /// ) + /// .await?; /// assert_eq!(reader.get(b"key").await?, Some("value".into())); /// Ok(()) /// } @@ -2248,9 +1647,14 @@ impl DbReader { /// ## Examples /// /// ``` - /// use slatedb::{Db, DbReader, DbReaderMode, config::DbReaderOptions, config::ReadOptions, Error}; - /// use slatedb::object_store::{ObjectStore, memory::InMemory}; - /// use std::sync::Arc; + /// use { + /// slatedb::{ + /// config::{DbReaderOptions, ReadOptions}, + /// object_store::{memory::InMemory, ObjectStore}, + /// Db, DbReader, DbReaderMode, Error, + /// }, + /// std::sync::Arc, + /// }; /// /// #[tokio::main] /// async fn main() -> Result<(), Error> { @@ -2260,12 +1664,16 @@ impl DbReader { /// db.flush().await?; /// /// let reader = DbReader::open( - /// "test_db", - /// Arc::clone(&object_store), - /// DbReaderMode::ManagedCheckpoint, - /// DbReaderOptions::default(), - /// ).await?; - /// assert_eq!(db.get_with_options(b"key", &ReadOptions::default()).await?, Some("value".into())); + /// "test_db", + /// Arc::clone(&object_store), + /// DbReaderMode::ManagedCheckpoint, + /// DbReaderOptions::default(), + /// ) + /// .await?; + /// assert_eq!( + /// db.get_with_options(b"key", &ReadOptions::default()).await?, + /// Some("value".into()) + /// ); /// Ok(()) /// } /// ``` @@ -2344,9 +1752,14 @@ impl DbReader { /// ## Examples /// /// ``` - /// use slatedb::{Db, DbReader, DbReaderMode, config::DbReaderOptions, Error}; - /// use slatedb::object_store::{ObjectStore, memory::InMemory}; - /// use std::sync::Arc; + /// use { + /// slatedb::{ + /// config::DbReaderOptions, + /// object_store::{memory::InMemory, ObjectStore}, + /// Db, DbReader, DbReaderMode, Error, + /// }, + /// std::sync::Arc, + /// }; /// /// #[tokio::main] /// async fn main() -> Result<(), Error> { @@ -2357,11 +1770,12 @@ impl DbReader { /// db.flush().await?; /// /// let reader = DbReader::open( - /// "test_db", - /// Arc::clone(&object_store), - /// DbReaderMode::ManagedCheckpoint, - /// DbReaderOptions::default(), - /// ).await?; + /// "test_db", + /// Arc::clone(&object_store), + /// DbReaderMode::ManagedCheckpoint, + /// DbReaderOptions::default(), + /// ) + /// .await?; /// let mut iter = reader.scan("a".."b").await?; /// let kv = iter.next().await?.unwrap(); /// assert_eq!(kv.key.as_ref(), b"a"); @@ -2449,8 +1863,8 @@ impl DbReader { /// /// ## Arguments /// - `prefix`: the key prefix to scan - /// - `subrange`: the range of key suffixes (relative to `prefix`) to - /// scan; `..` scans all keys with the prefix + /// - `subrange`: the range of key suffixes (relative to `prefix`) to scan; `..` scans all keys + /// with the prefix /// /// ## Returns /// - `Result`: An iterator with the results of the scan @@ -2473,8 +1887,8 @@ impl DbReader { /// /// ## Arguments /// - `prefix`: the key prefix to scan - /// - `subrange`: the range of key suffixes (relative to `prefix`) to - /// scan; `..` scans all keys with the prefix + /// - `subrange`: the range of key suffixes (relative to `prefix`) to scan; `..` scans all keys + /// with the prefix /// - `options`: the scan options to use /// /// ## Returns @@ -2505,9 +1919,14 @@ impl DbReader { /// ## Examples /// /// ``` - /// use slatedb::{Db, DbReader, DbReaderMode, config::DbReaderOptions, Error}; - /// use slatedb::object_store::{ObjectStore, memory::InMemory}; - /// use std::sync::Arc; + /// use { + /// slatedb::{ + /// config::DbReaderOptions, + /// object_store::{memory::InMemory, ObjectStore}, + /// Db, DbReader, DbReaderMode, Error, + /// }, + /// std::sync::Arc, + /// }; /// /// #[tokio::main] /// async fn main() -> Result<(), Error> { @@ -2519,22 +1938,22 @@ impl DbReader { /// object_store.clone(), /// DbReaderMode::ManagedCheckpoint, /// options, - /// ).await?; + /// ) + /// .await?; /// reader.close().await?; /// Ok(()) /// } /// ``` - /// pub async fn close(&self) -> Result<(), crate::Error> { // Fixed-checkpoint readers do not have a manifest poller to publish // their clean shutdown, so close the shared status explicitly for // both reader modes before shutting down any managed task. self.inner.status_manager.write_result(Ok(())); - self.inner.replay_tasks.cancel(); - let shutdown_result = self.task_executor.shutdown_task(DB_READER_TASK_NAME).await; - self.inner.replay_tasks.shutdown().await; - shutdown_result.map_err(Into::::into)?; + self.task_executor + .shutdown_task(DB_READER_TASK_NAME) + .await + .map_err(Into::::into)?; if let Err(e) = self.inner.table_store.close_cache().await { warn!("failed to close block cache [error={:?}]", e); @@ -2650,52 +2069,97 @@ impl DbCacheManagerOps for DbReader { #[cfg(test)] mod tests { - use super::{ - DbReaderMessage, ManifestPoller, ReaderGeneration, ReaderState, ReplayMemtables, - RuntimeMissingPolicy, MAX_RUNTIME_REPLAY_TURN_TIME, - }; - use crate::block_cache_policy::BlockCachePolicy; - use crate::clock::MonotonicClock; - use crate::config::{ - CheckpointOptions, CheckpointScope, FlushOptions, FlushType, MergeOptions, PutOptions, - Settings, WriteOptions, + use crate::wal::slatedb::reader::SlateDbWalReaderOptions; + use { + super::{ + DbReaderMessage, ManifestPoller, ReaderGeneration, ReaderState, ReplayMemtables, + WalReplayEnd, + }, + crate::{ + block_cache_policy::BlockCachePolicy, + clock::MonotonicClock, + config::{ + CheckpointOptions, CheckpointScope, CloseOptions, FlushOptions, FlushType, + MergeOptions, PutOptions, Settings, WriteOptions, + }, + db_cache::{test_utils::TestCache, DbCache}, + db_reader::{DbReader, DbReaderInner, DbReaderMode, DbReaderOptions}, + db_state::SstType, + db_stats::DbStats, + db_status::DbStatusManager, + dispatcher::MessageHandler, + error::SlateDBError, + format::sst::SsTableFormat, + iter::IterationOrder, + manifest::{ + store::{ManifestStore, StoredManifest}, + Manifest, ManifestCore, VersionedManifest, + }, + mem_table::{ImmutableMemtable, WritableKVTable}, + merge_operator::MergeOperatorType, + object_stores::ObjectStores, + oracle::DbReaderOracle, + paths::PathResolver, + proptest_util::{rng::new_test_rng, sample}, + reader::Reader, + tablestore::{TableStore, TableStoreKind}, + test_utils, + types::RowEntry, + wal::{ + slatedb::store::WalTableStore, WalError, WalFileRange, WalIterator, + WalReader as WalReaderTrait, WalRows, + }, + CloseReason, Db, + }, + bytes::Bytes, + fail_parallel::FailPointRegistry, + object_store::{memory::InMemory, path::Path, ObjectStore, ObjectStoreExt}, + rstest::rstest, + slatedb_common::{ + clock::{DefaultSystemClock, SystemClock}, + DbRand, MockSystemClock, + }, + std::{ + collections::{BTreeMap, BTreeSet, VecDeque}, + sync::{ + atomic::{AtomicUsize, Ordering}, + Arc, + }, + time::Duration, + }, + uuid::Uuid, }; - use crate::db_cache::test_utils::TestCache; - use crate::db_cache::DbCache; - use crate::db_reader::{DbReader, DbReaderInner, DbReaderMode, DbReaderOptions}; - use crate::db_state::SsTableId; - use crate::db_stats::DbStats; - use crate::db_status::DbStatusManager; - use crate::dispatcher::MessageHandler; - use crate::format::sst::SsTableFormat; - use crate::iter::{IterationOrder, RowEntryIterator}; - use crate::manifest::store::{ManifestStore, StoredManifest}; - use crate::manifest::{Manifest, ManifestCore, VersionedManifest}; - use crate::mem_table::{ImmutableMemtable, WritableKVTable}; - use crate::merge_operator::MergeOperatorType; - use crate::object_stores::ObjectStores; - use crate::oracle::DbReaderOracle; - use crate::paths::PathResolver; - use crate::proptest_util::rng::new_test_rng; - use crate::proptest_util::sample; - use crate::reader::Reader; - use crate::replay_task_scope::ReplayTaskScope; - use crate::tablestore::{TableStore, TableStoreKind}; - use crate::types::RowEntry; - use crate::{error::SlateDBError, test_utils, CloseReason, Db}; - use bytes::Bytes; - use fail_parallel::FailPointRegistry; - use object_store::memory::InMemory; - use object_store::path::Path; - use object_store::{ObjectStore, ObjectStoreExt}; - use rstest::rstest; - use slatedb_common::clock::{DefaultSystemClock, SystemClock}; - use slatedb_common::DbRand; - use slatedb_common::MockSystemClock; - use std::collections::{BTreeMap, VecDeque}; - use std::sync::Arc; - use std::time::Duration; - use uuid::Uuid; + + struct EmptyTestWalIterator; + + #[async_trait::async_trait] + impl WalIterator for EmptyTestWalIterator { + async fn next(&mut self) -> Result, WalError> { + Ok(None) + } + } + + #[derive(Default)] + struct CountingWalReader { + iterator_calls: AtomicUsize, + last_wal_file_id_calls: AtomicUsize, + } + + #[async_trait::async_trait] + impl WalReaderTrait for CountingWalReader { + async fn iterator( + &self, + _wal_file_id_range: WalFileRange, + ) -> Result, WalError> { + self.iterator_calls.fetch_add(1, Ordering::Relaxed); + Ok(Box::new(EmptyTestWalIterator)) + } + + async fn last_wal_file_id(&self, _replay_after_wal_id: u64) -> Result { + self.last_wal_file_id_calls.fetch_add(1, Ordering::Relaxed); + Ok(10) + } + } #[test] fn legacy_checkpoint_options_convert_to_reader_modes() { @@ -2730,24 +2194,6 @@ mod tests { .expect("reader did not install a new manifest generation"); } - async fn finish_manual_manifest_refresh(poller: &mut ManifestPoller) { - let receiver = poller - .tail_rx - .as_ref() - .expect("manual poller must retain its completion receiver") - .clone(); - loop { - let message = tokio::time::timeout(Duration::from_secs(5), receiver.recv()) - .await - .expect("manifest refresh did not complete") - .expect("manifest completion channel closed"); - poller.handle(message).await.unwrap(); - if !poller.manifest_refresh_in_flight { - break; - } - } - } - #[tokio::test] async fn should_get_value_from_db() { let object_store: Arc = Arc::new(InMemory::new()); @@ -2758,7 +2204,12 @@ mod tests { let key = b"test_key"; let value = b"test_value"; - db.put(key, value).await.unwrap(); + db.put(key, value) + .await + .unwrap() + .await_durable() + .await + .unwrap(); db.flush().await.unwrap(); let reader = DbReader::open( @@ -2776,6 +2227,32 @@ mod tests { ); } + #[tokio::test] + async fn db_reader_builder_should_use_custom_wal_reader() { + let object_store: Arc = Arc::new(InMemory::new()); + let path = Path::from("/tmp/test_custom_wal_reader"); + let db = Db::open(path.clone(), Arc::clone(&object_store)) + .await + .unwrap(); + db.close().await.unwrap(); + + let wal_reader = Arc::new(CountingWalReader::default()); + let reader = DbReader::builder(path, object_store) + .with_reader_mode(DbReaderMode::FollowLatest) + .with_wal_reader(wal_reader.clone()) + .with_options(DbReaderOptions { + manifest_poll_interval: Duration::from_secs(60 * 60), + ..DbReaderOptions::default() + }) + .build() + .await + .unwrap(); + + assert!(wal_reader.last_wal_file_id_calls.load(Ordering::Relaxed) > 0); + assert!(wal_reader.iterator_calls.load(Ordering::Relaxed) > 0); + reader.close().await.unwrap(); + } + #[tokio::test] async fn empty_database_reader_returns_typed_database_missing() { let object_store: Arc = Arc::new(InMemory::new()); @@ -2842,9 +2319,11 @@ mod tests { let reader = DbReader::open_internal( test_provider.manifest_store(), test_provider.table_store(), + test_provider.wal_store(), DbReaderMode::Checkpoint(checkpoint_result.id), None, None, + None, DbReaderOptions::default(), test_provider.system_clock.clone(), test_provider.rand.clone(), @@ -3050,7 +2529,7 @@ mod tests { let parent_manifest = Manifest::initial(ManifestCore::new()); let parent_path = "/tmp/parent_store".to_string(); - let source_checkpoint_id = uuid::Uuid::new_v4(); + let source_checkpoint_id = Uuid::new_v4(); let _ = StoredManifest::store_uninitialized_clone( Arc::clone(&manifest_store), @@ -3171,10 +2650,8 @@ mod tests { .contains("snapshots are unsupported in FollowLatest mode")); assert!(recording_store.write_kinds().is_empty()); - let (tail_tx, tail_rx) = async_channel::unbounded(); - let mut poller = ManifestPoller::new(Arc::clone(&reader.inner), tail_tx, tail_rx); + let mut poller = ManifestPoller::new(Arc::clone(&reader.inner)); poller.handle(DbReaderMessage::PollManifest).await.unwrap(); - finish_manual_manifest_refresh(&mut poller).await; assert!(reader.manifest().id() >= latest_manifest.id); assert_eq!( @@ -3213,9 +2690,11 @@ mod tests { let reader = DbReader::open_internal( test_provider.manifest_store(), test_provider.table_store(), + test_provider.wal_store(), DbReaderMode::FollowLatest, None, None, + None, DbReaderOptions { manifest_poll_interval: Duration::from_secs(60 * 60), ..DbReaderOptions::default() @@ -3243,10 +2722,8 @@ mod tests { saved_manifests.push((location, bytes)); } - let (tail_tx, tail_rx) = async_channel::unbounded(); - let mut poller = ManifestPoller::new(Arc::clone(&reader.inner), tail_tx, tail_rx); + let mut poller = ManifestPoller::new(Arc::clone(&reader.inner)); poller.handle(DbReaderMessage::PollManifest).await.unwrap(); - finish_manual_manifest_refresh(&mut poller).await; assert_eq!(reader.manifest().id(), manifest_id); assert_eq!( @@ -3267,7 +2744,6 @@ mod tests { db.close().await.unwrap(); poller.handle(DbReaderMessage::PollManifest).await.unwrap(); - finish_manual_manifest_refresh(&mut poller).await; assert!(reader.manifest().id() > manifest_id); assert_eq!( reader.get(b"key").await.unwrap(), @@ -3276,95 +2752,6 @@ mod tests { reader.close().await.unwrap(); } - #[tokio::test] - async fn close_cancels_a_blocked_manifest_get() { - let inner = Arc::new(InMemory::new()); - let writer_store: Arc = inner.clone(); - let gated = Arc::new(test_utils::GatedObjectStore::new(inner)); - let object_store: Arc = gated.clone(); - let path = Path::from("/tmp/test_reader_close_blocked_manifest_list"); - let provider = TestProvider::new(path.clone(), writer_store); - let db = provider.new_db(Settings::default()).await.unwrap(); - db.put(b"key", b"value").await.unwrap(); - db.flush().await.unwrap(); - - let reader = DbReader::open( - path, - object_store, - DbReaderMode::FollowLatest, - DbReaderOptions { - manifest_poll_interval: Duration::from_secs(60), - wal_poll_interval: Duration::from_secs(60), - checkpoint_lifetime: Duration::ZERO, - ..DbReaderOptions::default() - }, - ) - .await - .unwrap(); - - db.put(b"key", b"updated").await.unwrap(); - db.flush().await.unwrap(); - let (tx, rx) = async_channel::unbounded(); - let mut poller = ManifestPoller::new(Arc::clone(&reader.inner), tx, rx); - gated.get_opts_gate.close(); - poller.handle(DbReaderMessage::PollManifest).await.unwrap(); - tokio::time::timeout( - Duration::from_secs(1), - gated.get_opts_gate.wait_for_arrivals(1), - ) - .await - .expect("manifest refresh did not reach the blocked GET"); - - tokio::time::timeout(Duration::from_secs(1), reader.close()) - .await - .expect("reader close waited for a blocked manifest GET") - .unwrap(); - poller.cancel_manifest_refresh_task().await; - } - - #[tokio::test] - async fn close_cancels_a_blocked_checkpoint_manifest_list() { - let inner = Arc::new(InMemory::new()); - let writer_store: Arc = inner.clone(); - let gated = Arc::new(test_utils::GatedObjectStore::new(inner)); - let reader_store: Arc = gated.clone(); - let path = Path::from("/tmp/test_reader_close_blocked_checkpoint_list"); - let provider = TestProvider::new(path.clone(), writer_store); - let db = provider.new_db(Settings::default()).await.unwrap(); - db.put(b"key", b"value").await.unwrap(); - db.flush().await.unwrap(); - - let reader = DbReader::open( - path, - reader_store, - DbReaderMode::ManagedCheckpoint, - DbReaderOptions { - manifest_poll_interval: Duration::from_secs(60), - wal_poll_interval: Duration::from_secs(60), - ..DbReaderOptions::default() - }, - ) - .await - .unwrap(); - let (tx, rx) = async_channel::unbounded(); - let mut poller = ManifestPoller::new(Arc::clone(&reader.inner), tx, rx); - gated.list_gate.close(); - let poll = tokio::spawn(async move { poller.handle(DbReaderMessage::PollManifest).await }); - tokio::time::timeout(Duration::from_secs(1), gated.list_gate.wait_for_arrivals(1)) - .await - .expect("checkpoint refresh did not reach the blocked LIST"); - - tokio::time::timeout(Duration::from_secs(1), reader.close()) - .await - .expect("reader close waited for a blocked manifest LIST") - .unwrap(); - tokio::time::timeout(Duration::from_secs(1), poll) - .await - .expect("cancelled checkpoint poll remained blocked") - .unwrap() - .unwrap(); - } - #[tokio::test(start_paused = true)] async fn should_reestablish_reader_checkpoint() { let object_store: Arc = Arc::new(InMemory::new()); @@ -3378,7 +2765,6 @@ mod tests { let db = test_provider.new_db(db_options).await.unwrap(); let reader_options = DbReaderOptions { manifest_poll_interval: Duration::from_millis(10), - wal_poll_interval: Duration::from_millis(10), ..DbReaderOptions::default() }; let reader = test_provider @@ -3421,7 +2807,6 @@ mod tests { let _db = test_provider.new_db(Settings::default()).await; let reader_options = DbReaderOptions { manifest_poll_interval: Duration::from_millis(500), - wal_poll_interval: Duration::from_millis(10), checkpoint_lifetime: Duration::from_millis(1000), ..DbReaderOptions::default() }; @@ -3459,12 +2844,10 @@ mod tests { > initial_reader_checkpoint.expire_time.unwrap() ); - // Shutdown is a bounded local operation. It must not wait for a remote - // manifest write merely to delete the checkpoint; its finite lease is - // reclaimed by normal checkpoint GC after expiry. + // The checkpoint is removed on shutdown reader.close().await.unwrap(); let updated_manifest = manifest_store.read_latest_manifest().await.unwrap(); - assert_eq!(1, updated_manifest.manifest.core.checkpoints.len()); + assert_eq!(0, updated_manifest.manifest.core.checkpoints.len()); } // Regression test for https://github.com/slatedb/slatedb/issues/1750. @@ -3503,6 +2886,8 @@ mod tests { let inner = DbReaderInner::new( Arc::clone(&manifest_store), table_store, + test_provider.wal_store(), + None, DbReaderOptions { manifest_poll_interval: Duration::from_millis(100), checkpoint_lifetime: Duration::from_millis(1000), @@ -3598,6 +2983,8 @@ mod tests { let inner = DbReaderInner::new( Arc::clone(&manifest_store), table_store, + test_provider.wal_store(), + None, DbReaderOptions { manifest_poll_interval: Duration::from_millis(100), checkpoint_lifetime: Duration::from_millis(1000), @@ -3652,7 +3039,6 @@ mod tests { let reader_options = DbReaderOptions { manifest_poll_interval: Duration::from_millis(500), - wal_poll_interval: Duration::from_millis(10), checkpoint_lifetime: Duration::from_millis(1000), ..DbReaderOptions::default() }; @@ -3663,16 +3049,40 @@ mod tests { .unwrap(); let key = b"test_key"; let value = b"test_value"; - db.put(key, value).await.unwrap(); + db.put(key, value) + .await + .unwrap() + .await_durable() + .await + .unwrap(); db.flush().await.unwrap(); - tokio::time::sleep(Duration::from_millis(20)).await; + tokio::time::sleep(Duration::from_millis(500)).await; assert_eq!( reader.get(key).await.unwrap(), Some(Bytes::from_static(value)) ); } + #[tokio::test] + async fn default_durable_write_is_visible_to_reader_opened_immediately() { + let object_store: Arc = Arc::new(InMemory::new()); + let path = Path::from("/tmp/default_durable_write_is_visible_to_reader"); + let test_provider = TestProvider::new(path, Arc::clone(&object_store)); + let db = test_provider.new_db(Settings::default()).await.unwrap(); + + db.put(b"key", b"value").await.unwrap(); + + let reader = test_provider + .new_db_reader(DbReaderOptions::default(), None, None) + .await + .unwrap(); + assert_eq!( + reader.get(b"key").await.unwrap(), + Some(Bytes::from_static(b"value")) + ); + } + #[tokio::test] async fn reader_snapshot_should_remain_stable_while_wal_replay_advances() { let object_store: Arc = Arc::new(InMemory::new()); @@ -3690,7 +3100,7 @@ mod tests { let write_options = WriteOptions { await_durable: false, - ..WriteOptions::default() + ..Default::default() }; db.put_with_options(b"key", b"v1", &PutOptions::default(), &write_options) .await @@ -3777,7 +3187,7 @@ mod tests { .unwrap(); let write_options = WriteOptions { await_durable: false, - ..WriteOptions::default() + ..Default::default() }; db.put_with_options(b"key", b"v1", &PutOptions::default(), &write_options) .await @@ -3889,7 +3299,7 @@ mod tests { .unwrap(); let write_options = WriteOptions { await_durable: false, - ..WriteOptions::default() + ..Default::default() }; db.put_with_options(b"key", b"v1", &PutOptions::default(), &write_options) .await @@ -4049,7 +3459,7 @@ mod tests { .unwrap(); let write_options = WriteOptions { await_durable: false, - ..WriteOptions::default() + ..Default::default() }; db.put_with_options(b"key", b"v1", &PutOptions::default(), &write_options) .await @@ -4153,7 +3563,7 @@ mod tests { .unwrap(); let write_options = WriteOptions { await_durable: false, - ..WriteOptions::default() + ..Default::default() }; db.put_with_options(b"a", b"old", &PutOptions::default(), &write_options) .await @@ -4255,16 +3665,17 @@ mod tests { let path = Path::from("/tmp/test_db_reader_replay_order"); let test_provider = TestProvider::new(path, Arc::clone(&object_store)); let table_store = test_provider.table_store(); + let wal_store = test_provider.wal_store(); write_wal_sst( - Arc::clone(&table_store), + Arc::clone(&wal_store), 3, vec![RowEntry::new_value(b"stale_key", b"stale_value", 3)], ) .await .unwrap(); write_wal_sst( - Arc::clone(&table_store), + Arc::clone(&wal_store), 4, vec![RowEntry::new_value(b"fresh_key", b"fresh_value", 4)], ) @@ -4286,409 +3697,118 @@ mod tests { let mut core = ManifestCore::new(); core.next_wal_sst_id = 5; + let status_manager = status_manager_for_core(&core); - let (last_wal_id, last_committed_seq) = DbReaderInner::replay_wal_into_exact( - Arc::clone(&table_store), - &DbReaderOptions::default(), - &core, - &mut into_tables, - None, - false, - None, - None, - None, - ) - .await - .unwrap(); - - assert_eq!(last_wal_id, 4); - assert_eq!(last_committed_seq, 4); - - let newest_replayed = into_tables.front().unwrap(); - assert_eq!(newest_replayed.recent_flushed_wal_id(), 4); - - let newest_table = newest_replayed.table(); - let mut newest_iter = newest_table.iter(); - test_utils::assert_iterator( - &mut newest_iter, - vec![RowEntry::new_value(b"fresh_key", b"fresh_value", 4)], - ) - .await; - } - - #[test] - fn replay_memtable_prepend_should_share_the_existing_tail() { - let mut original = ReplayMemtables::default(); - original.prepend(immutable_memtable( - 1, - vec![RowEntry::new_value(b"old", b"value", 1)], - )); - let original_head = Arc::clone(original.head.as_ref().unwrap()); - - let mut extended = original.clone(); - extended.prepend(immutable_memtable( - 2, - vec![RowEntry::new_value(b"new", b"value", 2)], - )); - - let shared_tail = extended.head.as_ref().unwrap().older.as_ref().unwrap(); - assert!(Arc::ptr_eq(shared_tail, &original_head)); - assert_eq!(1, original.len()); - assert_eq!(2, extended.len()); - } - - #[tokio::test] - async fn replay_wal_into_should_publish_each_bounded_batch() { - let object_store: Arc = Arc::new(InMemory::new()); - let path = Path::from("/tmp/test_db_reader_incremental_replay_publication"); - let test_provider = TestProvider::new(path, Arc::clone(&object_store)); - let table_store = test_provider.table_store(); - let rows = (1..=3) - .map(|seq| RowEntry::new_value(format!("key-{seq}").as_bytes(), &[b'x'; 128], seq)) - .collect::>(); - for (wal_id, row) in rows.iter().cloned().enumerate() { - write_wal_sst(Arc::clone(&table_store), wal_id as u64 + 1, vec![row]) - .await - .unwrap(); - } - let max_memtable_bytes = - table_store.estimate_encoded_size_compacted(1, rows[0].estimated_size()) as u64; - let options = DbReaderOptions { - max_memtable_bytes, - ..DbReaderOptions::default() - }; - let mut core = ManifestCore::new(); - core.next_wal_sst_id = 4; - let mut into_tables = ReplayMemtables::default(); - let mut publications = Vec::new(); - let mut publish = |tables: &ReplayMemtables, wal_id, seq| { - publications.push((wal_id, seq, tables.len())); - }; - - let result = DbReaderInner::replay_wal_into_exact( - Arc::clone(&table_store), - &options, - &core, - &mut into_tables, - None, - false, - None, - Some(&mut publish), - None, - ) - .await - .unwrap(); - - assert_eq!((3, 3), result); - assert_eq!(vec![(1, 1, 1), (2, 2, 2), (3, 3, 3)], publications); - } - - #[tokio::test] - async fn runtime_known_missing_wal_is_error_and_exact_next_probe_reports_caught_up() { - let object_store: Arc = Arc::new(InMemory::new()); - let path = Path::from("/tmp/test_db_reader_runtime_missing_contract"); - let table_store = TestProvider::new(path, object_store).table_store(); - let options = DbReaderOptions::default(); - let mut core = ManifestCore::new(); - core.next_wal_sst_id = 2; - let mut known_cursor = (0, 0); - - let known_missing = DbReaderInner::replay_runtime_range_into( - Arc::clone(&table_store), - &options, - &core, - &mut ReplayMemtables::default(), - &mut known_cursor, - 1..2, - RuntimeMissingPolicy::Error, - None, - None, - None, - None, - ) - .await; - assert!(known_missing.unwrap_err().has_object_store_not_found()); - - let mut tail_cursor = (0, 0); - let caught_up = DbReaderInner::replay_runtime_range_into( - table_store, - &options, - &core, - &mut ReplayMemtables::default(), - &mut tail_cursor, - 1..2, - RuntimeMissingPolicy::ExactNextIsCaughtUp(1), - None, - None, - None, - None, - ) - .await - .unwrap(); - assert!(caught_up); - assert_eq!(tail_cursor, (0, 0)); - } - - #[tokio::test] - async fn manifest_replay_yields_at_the_sixty_four_wal_boundary() { - let object_store: Arc = Arc::new(InMemory::new()); - let path = Path::from("/tmp/test_db_reader_manifest_turn_limit"); - let table_store = TestProvider::new(path, object_store).table_store(); - for wal_id in 1..=65 { - write_wal_sst( - Arc::clone(&table_store), - wal_id, - vec![RowEntry::new_value( - format!("key-{wal_id:03}").as_bytes(), - b"value", - wal_id, - )], - ) - .await - .unwrap(); - } - let mut core = ManifestCore::new(); - core.next_wal_sst_id = 66; - let mut replayed = ReplayMemtables::default(); - let clock: Arc = Arc::new(DefaultSystemClock::new()); - let scope = ReplayTaskScope::new(); - - let cursor = DbReaderInner::replay_manifest_range_into( - table_store, - &DbReaderOptions::default(), - &core, - &mut replayed, - None, - None, - None, - Some(scope.clone()), - &clock, - ) - .await - .unwrap(); - - assert_eq!(cursor, (65, 65)); - assert_eq!(replayed.len(), 2); - scope.shutdown().await; - } - - #[tokio::test] - async fn runtime_tail_yields_after_sixty_four_wals_and_continues_without_listing() { - let recording = Arc::new(test_utils::RecordingObjectStore::new(Arc::new( - InMemory::new(), - ))); - let object_store: Arc = recording.clone(); - let path = Path::from("/tmp/test_db_reader_runtime_turn_limit"); - let provider = TestProvider::new(path, object_store); - let db = provider - .new_db(Settings { - flush_interval: None, - compactor_options: None, - garbage_collector_options: None, - ..Settings::default() - }) - .await - .unwrap(); - let checkpoint = db - .create_checkpoint(CheckpointScope::All, &CheckpointOptions::default()) - .await - .unwrap(); - db.close().await.unwrap(); - let reader = provider - .new_db_reader(DbReaderOptions::default(), Some(checkpoint.id), None) - .await - .unwrap(); - let table_store = provider.table_store(); - let initial_wal_id = reader.inner.state.read().last_wal_id; - for wal_id in initial_wal_id + 1..=initial_wal_id + 65 { - write_wal_sst( - Arc::clone(&table_store), - wal_id, - vec![RowEntry::new_value( - format!("key-{wal_id:03}").as_bytes(), - b"value", - wal_id, - )], - ) - .await - .unwrap(); - } - recording.clear(); - - let first_base = Arc::clone(&reader.inner.state.read()); - let first = DbReaderInner::replay_runtime_turn( - Arc::clone(&reader.inner), - first_base, - reader.inner.replay_tasks.child(), - ) - .await - .unwrap(); - assert_eq!(first.last_wal_id, initial_wal_id + 64); - assert!(first.reached_turn_limit); - assert!(!first.caught_up); - - let second_base = Arc::new(ReaderState { - generation: Arc::clone(&reader.inner.state.read().generation), - imm_memtable: first.imm_memtable, - last_wal_id: first.last_wal_id, - last_remote_persisted_seq: first.last_committed_seq, - }); - let second = DbReaderInner::replay_runtime_turn( - Arc::clone(&reader.inner), - second_base, - reader.inner.replay_tasks.child(), - ) - .await - .unwrap(); - assert_eq!(second.last_wal_id, initial_wal_id + 65); - assert!(second.caught_up); - assert!(!second.reached_turn_limit); - assert_eq!(recording.list_calls(), 0); - - reader.close().await.unwrap(); - } - - #[tokio::test(start_paused = true)] - async fn runtime_tail_deadline_wins_without_publishing_a_partial_wal() { - let inner_store: Arc = Arc::new(InMemory::new()); - let gated = Arc::new(test_utils::GatedObjectStore::new(inner_store)); - let object_store: Arc = gated.clone(); - let path = Path::from("/tmp/test_db_reader_runtime_deadline"); - let provider = TestProvider::new(path, object_store); - let db = provider - .new_db(Settings { - flush_interval: None, - compactor_options: None, - garbage_collector_options: None, - ..Settings::default() - }) - .await - .unwrap(); - let reader = provider - .new_db_reader( - DbReaderOptions { - // Keep the test-owned replay turn isolated from the background poller. - skip_wal_replay: true, - ..DbReaderOptions::default() - }, - None, - None, - ) - .await - .unwrap(); - let base = Arc::clone(&reader.inner.state.read()); - let base_memtable_count = base.imm_memtable.len(); - db.put_with_options( - b"deadline-key", - b"deadline-value", - &PutOptions::default(), - &WriteOptions { - await_durable: false, - ..WriteOptions::default() - }, - ) - .await - .unwrap(); - db.flush_with_options(FlushOptions { - flush_type: FlushType::Wal, - }) - .await - .unwrap(); - - let prior_head_arrivals = gated.head_gate.arrivals(); - gated.head_gate.close(); - let blocked = tokio::spawn(DbReaderInner::replay_runtime_turn( - Arc::clone(&reader.inner), - Arc::clone(&base), - reader.inner.replay_tasks.child(), - )); - gated - .head_gate - .wait_for_arrivals(prior_head_arrivals + 1) - .await; - tokio::time::advance(MAX_RUNTIME_REPLAY_TURN_TIME).await; - // Make the object read ready at the exact deadline. The deadline branch - // is biased and must still win. - gated.head_gate.release(); - let expired = blocked.await.unwrap().unwrap(); - assert_eq!(expired.last_wal_id, base.last_wal_id); - assert_eq!(expired.last_committed_seq, base.last_remote_persisted_seq); - assert_eq!(expired.imm_memtable.len(), base_memtable_count); - assert!(expired.reached_turn_limit); - assert!(!expired.caught_up); - assert_eq!(reader.inner.state.read().last_wal_id, base.last_wal_id); - - let completed = DbReaderInner::replay_runtime_turn( - Arc::clone(&reader.inner), - base, - reader.inner.replay_tasks.child(), + let (last_wal_id, last_committed_seq) = DbReaderInner::replay_wal_into( + Arc::clone(&table_store), + &native_wal_reader(&wal_store, &status_manager), + &DbReaderOptions::default(), + &core, + &mut into_tables, + None, + WalReplayEnd::Manifest, + None, + None, + None, ) .await .unwrap(); - assert!(completed.last_wal_id > expired.last_wal_id); - assert!(completed.caught_up); - let mut rows = completed - .imm_memtable - .front() - .expect("completed WAL must be present") - .table() - .iter(); - let row = rows.next().await.unwrap().unwrap(); - assert_eq!(row.key, Bytes::from_static(b"deadline-key")); - assert_eq!( - row.value, - crate::types::ValueDeletable::Value(Bytes::from_static(b"deadline-value")) - ); - reader.close().await.unwrap(); - db.close().await.unwrap(); + assert_eq!(last_wal_id, 4); + assert_eq!(last_committed_seq, 4); + + let newest_replayed = into_tables.front().unwrap(); + assert_eq!(newest_replayed.recent_flushed_wal_id(), 4); + + let newest_table = newest_replayed.table(); + let mut newest_iter = newest_table.iter(); + test_utils::assert_iterator( + &mut newest_iter, + vec![RowEntry::new_value(b"fresh_key", b"fresh_value", 4)], + ) + .await; } #[test] - fn reader_validation_rejects_invalid_wal_replay_settings_in_every_mode() { - for mode in [ - DbReaderMode::Checkpoint(Uuid::nil()), - DbReaderMode::ManagedCheckpoint, - DbReaderMode::FollowLatest, - ] { - let options = DbReaderOptions { - wal_replay: crate::config::WalReplaySettings { - max_concurrent_objects: 0, - ..Default::default() - }, - ..Default::default() - }; - assert!(DbReader::validate_options(mode, &options).is_err()); - } + fn replay_memtable_prepend_should_share_the_existing_tail() { + let mut original = ReplayMemtables::default(); + original.prepend(immutable_memtable( + 1, + vec![RowEntry::new_value(b"old", b"value", 1)], + )); + let original_head = Arc::clone(original.head.as_ref().unwrap()); + + let mut extended = original.clone(); + extended.prepend(immutable_memtable( + 2, + vec![RowEntry::new_value(b"new", b"value", 2)], + )); + + let shared_tail = extended.head.as_ref().unwrap().older.as_ref().unwrap(); + assert!(Arc::ptr_eq(shared_tail, &original_head)); + assert_eq!(1, original.len()); + assert_eq!(2, extended.len()); } - #[test] - fn reader_validation_rejects_zero_wal_poll_interval_in_every_mode() { - for mode in [ - DbReaderMode::Checkpoint(Uuid::nil()), - DbReaderMode::ManagedCheckpoint, - DbReaderMode::FollowLatest, - ] { - let options = DbReaderOptions { - wal_poll_interval: Duration::ZERO, - ..DbReaderOptions::default() - }; - assert!(matches!( - DbReader::validate_options(mode, &options), - Err(SlateDBError::InvalidWalPollInterval(interval)) if interval.is_zero() - )); + #[tokio::test] + async fn replay_wal_into_should_publish_each_bounded_batch() { + let object_store: Arc = Arc::new(InMemory::new()); + let path = Path::from("/tmp/test_db_reader_incremental_replay_publication"); + let test_provider = TestProvider::new(path, Arc::clone(&object_store)); + let table_store = test_provider.table_store(); + let wal_store = test_provider.wal_store(); + let rows = (1..=3) + .map(|seq| RowEntry::new_value(format!("key-{seq}").as_bytes(), &[b'x'; 128], seq)) + .collect::>(); + for (wal_id, row) in rows.iter().cloned().enumerate() { + write_wal_sst(Arc::clone(&wal_store), wal_id as u64 + 1, vec![row]) + .await + .unwrap(); } + let max_memtable_bytes = + table_store.estimate_encoded_size_compacted(1, rows[0].estimated_size()) as u64; + let options = DbReaderOptions { + max_memtable_bytes, + ..DbReaderOptions::default() + }; + let mut core = ManifestCore::new(); + core.next_wal_sst_id = 4; + let status_manager = status_manager_for_core(&core); + let mut into_tables = ReplayMemtables::default(); + let mut publications = Vec::new(); + let mut publish = |tables: &ReplayMemtables, wal_id, seq| { + publications.push((wal_id, seq, tables.len())); + }; + + let result = DbReaderInner::replay_wal_into( + Arc::clone(&table_store), + &native_wal_reader(&wal_store, &status_manager), + &options, + &core, + &mut into_tables, + None, + WalReplayEnd::Manifest, + None, + Some(&mut publish), + None, + ) + .await + .unwrap(); + + assert_eq!((3, 3), result); + assert_eq!(vec![(1, 1, 1), (2, 2, 2), (3, 3, 3)], publications); } #[tokio::test] - async fn exact_replay_should_fail_closed_on_a_missing_wal() { + async fn replay_wal_into_should_treat_missing_wal_sst_as_end_of_iteration() { let object_store: Arc = Arc::new(InMemory::new()); let path = Path::from("/tmp/test_db_reader_missing_wal"); let test_provider = TestProvider::new(path, Arc::clone(&object_store)); let table_store = test_provider.table_store(); + let wal_store = test_provider.wal_store(); write_wal_sst( - Arc::clone(&table_store), + Arc::clone(&wal_store), 1, vec![RowEntry::new_value(b"key", b"value", 1)], ) @@ -4698,31 +3818,38 @@ mod tests { let mut into_tables = ReplayMemtables::default(); let mut core = ManifestCore::new(); core.next_wal_sst_id = 3; + let status_manager = status_manager_for_core(&core); - let error = DbReaderInner::replay_wal_into_exact( + let (last_wal_id, last_committed_seq) = DbReaderInner::replay_wal_into( Arc::clone(&table_store), + &native_wal_reader(&wal_store, &status_manager), &DbReaderOptions::default(), &core, &mut into_tables, None, - false, + WalReplayEnd::Manifest, None, None, None, ) .await - .unwrap_err(); + .unwrap(); - assert!(error.has_object_store_not_found()); - assert!(into_tables.is_empty()); + // WAL 2 is missing and ends the iteration, but the rows already replayed + // from WAL 1 must still be returned. + assert_eq!(last_wal_id, 1); + assert_eq!(last_committed_seq, 1); + assert_eq!(into_tables.len(), 1); + assert_eq!(into_tables.front().unwrap().recent_flushed_wal_id(), 1); } #[tokio::test] - async fn exact_replay_should_surface_a_later_missing_wal() { + async fn replay_wal_into_should_keep_previously_replayed_tables_before_missing_wal_sst() { let object_store: Arc = Arc::new(InMemory::new()); let path = Path::from("/tmp/test_db_reader_missing_wal_after_replay"); let test_provider = TestProvider::new(path, Arc::clone(&object_store)); let table_store = test_provider.table_store(); + let wal_store = test_provider.wal_store(); let wal_1_row = RowEntry::new_value(b"a", &[b'a'; 8], 1); let wal_2_row_1 = RowEntry::new_value(b"b", &[b'b'; 40], 2); @@ -4733,11 +3860,11 @@ mod tests { wal_1_row.estimated_size() + wal_2_row_1.estimated_size(), ) as u64; - write_wal_sst(Arc::clone(&table_store), 1, vec![wal_1_row.clone()]) + write_wal_sst(Arc::clone(&wal_store), 1, vec![wal_1_row.clone()]) .await .unwrap(); write_wal_sst( - Arc::clone(&table_store), + Arc::clone(&wal_store), 2, vec![wal_2_row_1.clone(), wal_2_row_2.clone()], ) @@ -4752,22 +3879,25 @@ mod tests { max_memtable_bytes, ..DbReaderOptions::default() }; + let status_manager = status_manager_for_core(&core); - let error = DbReaderInner::replay_wal_into_exact( + let (last_wal_id, last_committed_seq) = DbReaderInner::replay_wal_into( Arc::clone(&table_store), + &native_wal_reader(&wal_store, &status_manager), &reader_options, &core, &mut into_tables, None, - false, + WalReplayEnd::Manifest, None, None, None, ) .await - .unwrap_err(); + .unwrap(); - assert!(error.has_object_store_not_found()); + assert_eq!(last_wal_id, 2); + assert_eq!(last_committed_seq, 3); assert_eq!(into_tables.len(), 1); let replayed = into_tables.front().unwrap(); @@ -4787,17 +3917,20 @@ mod tests { let path = Path::from("/tmp/test_db_reader_fresh_db_no_writes"); let test_provider = TestProvider::new(path, Arc::clone(&object_store)); let table_store = test_provider.table_store(); + let wal_store = test_provider.wal_store(); let mut into_tables = ReplayMemtables::default(); let core = ManifestCore::new(); + let status_manager = status_manager_for_core(&core); - let (last_wal_id, last_committed_seq) = DbReaderInner::replay_wal_into_exact( + let (last_wal_id, last_committed_seq) = DbReaderInner::replay_wal_into( Arc::clone(&table_store), + &native_wal_reader(&wal_store, &status_manager), &DbReaderOptions::default(), &core, &mut into_tables, None, - true, + WalReplayEnd::Latest, None, None, None, @@ -4816,22 +3949,25 @@ mod tests { let path = Path::from("/tmp/test_db_reader_fresh_db_one_wal"); let test_provider = TestProvider::new(path, Arc::clone(&object_store)); let table_store = test_provider.table_store(); + let wal_store = test_provider.wal_store(); let wal_row = RowEntry::new_value(b"key", b"value", 1); - write_wal_sst(Arc::clone(&table_store), 1, vec![wal_row.clone()]) + write_wal_sst(Arc::clone(&wal_store), 1, vec![wal_row.clone()]) .await .unwrap(); let mut into_tables = ReplayMemtables::default(); let core = ManifestCore::new(); + let status_manager = status_manager_for_core(&core); - let (last_wal_id, last_committed_seq) = DbReaderInner::replay_wal_into_exact( + let (last_wal_id, last_committed_seq) = DbReaderInner::replay_wal_into( Arc::clone(&table_store), + &native_wal_reader(&wal_store, &status_manager), &DbReaderOptions::default(), &core, &mut into_tables, None, - true, + WalReplayEnd::Latest, None, None, None, @@ -4856,8 +3992,9 @@ mod tests { let path = Path::from("/tmp/test_db_reader_empty_fence_wal"); let test_provider = TestProvider::new(path, Arc::clone(&object_store)); let table_store = test_provider.table_store(); + let wal_store = test_provider.wal_store(); - write_wal_sst(Arc::clone(&table_store), 6, vec![]) + write_wal_sst(Arc::clone(&wal_store), 6, vec![]) .await .unwrap(); @@ -4873,14 +4010,16 @@ mod tests { let mut core = ManifestCore::new(); core.last_l0_seq = 8; core.next_wal_sst_id = 5; + let status_manager = status_manager_for_core(&core); - let (last_wal_id, last_committed_seq) = DbReaderInner::replay_wal_into_exact( + let (last_wal_id, last_committed_seq) = DbReaderInner::replay_wal_into( Arc::clone(&table_store), + &native_wal_reader(&wal_store, &status_manager), &DbReaderOptions::default(), &core, &mut into_tables, None, - true, + WalReplayEnd::Latest, None, None, None, @@ -4892,13 +4031,14 @@ mod tests { assert_eq!(last_committed_seq, 10); let head_after_first_replay = Arc::clone(into_tables.head.as_ref().unwrap()); - let (last_wal_id, last_committed_seq) = DbReaderInner::replay_wal_into_exact( + let (last_wal_id, last_committed_seq) = DbReaderInner::replay_wal_into( Arc::clone(&table_store), + &native_wal_reader(&wal_store, &status_manager), &DbReaderOptions::default(), &core, &mut into_tables, Some((last_wal_id, last_committed_seq)), - true, + WalReplayEnd::Latest, None, None, None, @@ -4927,16 +4067,17 @@ mod tests { let reader_options = DbReaderOptions { manifest_poll_interval: Duration::from_millis(500), - wal_poll_interval: Duration::from_millis(10), ..DbReaderOptions::default() }; let metrics_recorder = Arc::new(DefaultMetricsRecorder::new()); let reader = DbReader::open_internal( test_provider.manifest_store(), test_provider.table_store(), + test_provider.wal_store(), DbReaderMode::ManagedCheckpoint, None, None, + None, reader_options, Arc::clone(&test_provider.system_clock), Arc::clone(&test_provider.rand), @@ -4954,15 +4095,15 @@ mod tests { fail_parallel::cfg( Arc::clone(&test_provider.fp_registry), - "runtime-wal-replay", + "probe-wal-ssts", "return", ) .unwrap(); - tokio::time::sleep(Duration::from_millis(30)).await; + tokio::time::sleep(Duration::from_millis(20)).await; let result = reader.get(b"key").await.unwrap_err(); assert_eq!( result.to_string(), - "Unavailable error: io error (runtime WAL replay failpoint)" + "Unavailable error: wal unavailable (io error)" ); assert_eq!( Some(0), @@ -5012,12 +4153,6 @@ mod tests { .new_db_reader(reader_options.clone(), None, None) .await .unwrap(); - let (tail_tx, tail_rx) = async_channel::unbounded(); - let mut skipped_poller = ManifestPoller::new(Arc::clone(&reader.inner), tail_tx, tail_rx); - assert!(!skipped_poller.runtime_tailing_enabled()); - assert_eq!(skipped_poller.tickers().len(), 1); - skipped_poller.start_runtime_tail(); - assert!(!skipped_poller.tail_in_flight); // Should see the L0 flushed data assert_eq!( @@ -5094,9 +4229,9 @@ mod tests { // Inject a failpoint on WAL probing before flushing so it is active // when the poller fires. With the buggy replay_new_wals=true, - // reestablish_checkpoint calls last_seen_wal_id() which probes WAL SSTs + // reestablish_checkpoint resolves the last WAL file by probing WAL SSTs // and hits this failpoint. With the fix (replay_new_wals=false), the - // WAL probing is skipped entirely. + // WAL probe is skipped entirely. fail_parallel::cfg( Arc::clone(&test_provider.fp_registry), "probe-wal-ssts", @@ -5117,7 +4252,7 @@ mod tests { // Wait for the manifest poller to see the changed L0 state and // reestablish the checkpoint. Without the fix, the poller crashes - // on the WAL listing failpoint. + // on the WAL probing failpoint. let timeout = Duration::from_secs(5); let start = tokio::time::Instant::now(); loop { @@ -5175,6 +4310,204 @@ mod tests { ); } + /// A manifest records the WAL files written since the last L0 flush in + /// `next_wal_sst_id`. Opening a reader must not read them when WAL replay + /// is skipped: that range is exactly the expensive one, since it grows + /// with everything written between L0 flushes. + #[tokio::test] + async fn skip_wal_replay_should_not_read_wals_recorded_in_manifest() { + let recording_store = Arc::new(test_utils::RecordingObjectStore::new(Arc::new( + InMemory::new(), + ))); + let object_store: Arc = recording_store.clone(); + let path = Path::from("/tmp/test_kv_store"); + let test_provider = TestProvider::new(path.clone(), Arc::clone(&object_store)); + + let db = test_provider.new_db(Settings::default()).await.unwrap(); + let flushed_key = b"flushed_key"; + let flushed_value = b"flushed_value"; + db.put(flushed_key, flushed_value).await.unwrap(); + db.flush_with_options(FlushOptions { + flush_type: FlushType::MemTable, + }) + .await + .unwrap(); + + // Write data that stays in the WAL, then close without flushing the + // memtable. Closing persists the manifest, so `next_wal_sst_id` covers + // these WAL files while `replay_after_wal_id` stays at the last L0 flush. + // The write must be awaited to durability first: closing without a + // memtable flush does not flush the WAL, so an unawaited write would + // race the flush interval and might never reach a WAL SST. + let wal_only_key = b"wal_only_key"; + db.put(wal_only_key, b"wal_only_value").await.unwrap(); + db.close_with_options(CloseOptions::default().with_flush_type(Some(FlushType::Wal))) + .await + .unwrap(); + + let core = test_provider + .manifest_store() + .read_latest_manifest() + .await + .unwrap() + .manifest + .core; + assert!( + core.replay_after_wal_id + 1 < core.next_wal_sst_id, + "test needs a manifest that records live WAL files \ + [replay_after_wal_id={}, next_wal_sst_id={}]", + core.replay_after_wal_id, + core.next_wal_sst_id + ); + + recording_store.clear(); + let reader = test_provider + .new_db_reader( + DbReaderOptions { + skip_wal_replay: true, + ..DbReaderOptions::default() + }, + None, + None, + ) + .await + .unwrap(); + + let wal_reads = recording_store + .get_sst_types(false) + .into_iter() + .chain(recording_store.get_sst_types(true)) + .filter(|sst_type| *sst_type == Some(SstType::Wal)) + .count(); + assert_eq!(wal_reads, 0, "reader read WAL SSTs despite skip_wal_replay"); + + assert_eq!(reader.get(wal_only_key).await.unwrap(), None); + assert_eq!( + reader.get(flushed_key).await.unwrap(), + Some(Bytes::from_static(flushed_value)) + ); + } + + /// Regression test for #2003: read-ahead was a fixed 1MiB, so a large WAL took one + /// GET per MiB. It now covers a whole WAL SST, so reads for one file stay a small + /// constant instead of growing with size. + #[tokio::test] + async fn replay_reads_a_large_wal_sst_in_a_bounded_number_of_requests() { + let recording_store = Arc::new(test_utils::RecordingObjectStore::new(Arc::new( + InMemory::new(), + ))); + let object_store: Arc = recording_store.clone(); + let path = Path::from("/tmp/test_kv_store"); + let test_provider = TestProvider::new(path.clone(), Arc::clone(&object_store)); + let wal_store = test_provider.wal_store(); + + // One 16MiB WAL SST, far over the old 1MiB window. The old window read it in + // ~16 data GETs; the fix reads it in one. + let value = vec![b'x'; 4096]; + let entries: Vec = (0..4096u32) + .map(|i| RowEntry::new_value(format!("key-{i:08}").as_bytes(), &value, i as u64 + 1)) + .collect(); + write_wal_sst(Arc::clone(&wal_store), 1, entries) + .await + .unwrap(); + + let mut core = ManifestCore::new(); + core.next_wal_sst_id = 2; + let status_manager = status_manager_for_core(&core); + let wal_reader = native_wal_reader(&wal_store, &status_manager); + + recording_store.clear(); + let mut iterator = wal_reader.iterator((1..2).into()).await.unwrap(); + let mut rows = 0; + while let Some(batch) = iterator.next().await.unwrap() { + rows += batch.rows.len(); + } + assert_eq!(rows, 4096, "replay should return every WAL row"); + + let wal_reads = recording_store + .get_sst_types(false) + .into_iter() + .filter(|sst_type| *sst_type == Some(SstType::Wal)) + .count(); + // Footer, index, and one data read. The old 1MiB window needed ~16 data reads + // for this file, so the bound is the regression. + assert!( + wal_reads <= 4, + "expected a bounded number of WAL reads for one file, got {wal_reads}" + ); + } + + /// A checkpoint captures the WAL files that were durable when it was taken, + /// so a pinned reader replays them by default. `skip_wal_replay` opts out of + /// that read, at the cost of not seeing the checkpointed WAL writes. + #[rstest] + #[case(true, None)] + #[case(false, Some(Bytes::from_static(b"wal_only_value")))] + #[tokio::test] + async fn skip_wal_replay_should_control_checkpoint_wal_reads( + #[case] skip_wal_replay: bool, + #[case] expected: Option, + ) { + let object_store: Arc = Arc::new(InMemory::new()); + let path = Path::from("/tmp/test_kv_store"); + let test_provider = TestProvider::new(path.clone(), Arc::clone(&object_store)); + + let db = test_provider.new_db(Settings::default()).await.unwrap(); + db.put(b"flushed_key", b"flushed_value").await.unwrap(); + db.flush_with_options(FlushOptions { + flush_type: FlushType::MemTable, + }) + .await + .unwrap(); + + // This write is only durable in the WAL, so the checkpoint references it + // through the manifest's `next_wal_sst_id` rather than through L0. The + // scope must be `Durable`: `All` would flush the memtable to L0 first, + // leaving the checkpoint with no live WAL. + let wal_only_key = b"wal_only_key"; + db.put(wal_only_key, b"wal_only_value").await.unwrap(); + db.flush_with_options(FlushOptions { + flush_type: FlushType::Wal, + }) + .await + .unwrap(); + let checkpoint = db + .create_checkpoint(CheckpointScope::Durable, &CheckpointOptions::default()) + .await + .unwrap(); + db.close_with_options(CloseOptions::default().with_flush_type(Some(FlushType::Wal))) + .await + .unwrap(); + + let core = test_provider + .manifest_store() + .read_manifest(checkpoint.manifest_id) + .await + .unwrap() + .core; + assert!( + core.replay_after_wal_id + 1 < core.next_wal_sst_id, + "test needs a checkpoint whose manifest records live WAL files \ + [replay_after_wal_id={}, next_wal_sst_id={}]", + core.replay_after_wal_id, + core.next_wal_sst_id + ); + + let reader = test_provider + .new_db_reader( + DbReaderOptions { + skip_wal_replay, + ..DbReaderOptions::default() + }, + Some(checkpoint.id), + None, + ) + .await + .unwrap(); + + assert_eq!(reader.get(wal_only_key).await.unwrap(), expected); + } + struct TestProvider { object_store: Arc, path: Path, @@ -5215,7 +4548,9 @@ mod tests { DbReader::open_internal( self.manifest_store(), self.table_store(), + self.wal_store(), mode, + None, merge_operator, None, options, @@ -5227,6 +4562,26 @@ mod tests { } } + fn status_manager_for_core(core: &ManifestCore) -> DbStatusManager { + DbStatusManager::new_with_initial_values( + core.last_l0_seq, + VersionedManifest::from_manifest(1, Manifest::initial(core.clone())), + BTreeSet::default(), + ) + } + + fn native_wal_reader( + wal_store: &Arc, + status_manager: &DbStatusManager, + ) -> crate::wal::slatedb::reader::SlateDbWalReader { + crate::wal::slatedb::reader::SlateDbWalReader::new_with_status_manager( + Arc::clone(wal_store), + status_manager, + Arc::new(DefaultSystemClock::new()), + SlateDbWalReaderOptions::default(), + ) + } + fn immutable_memtable( recent_flushed_wal_id: u64, entries: Vec, @@ -5239,15 +4594,16 @@ mod tests { } async fn write_wal_sst( - table_store: Arc, + wal_store: Arc, wal_id: u64, entries: Vec, ) -> Result<(), SlateDBError> { - let mut writer = table_store.table_writer(SsTableId::Wal(wal_id)); + let mut builder = wal_store.table_builder(); for entry in entries { - writer.add(entry).await?; + builder.add(entry).await?; } - writer.close().await?; + let encoded_sst = builder.build().await?; + wal_store.write_sst(wal_id, &encoded_sst).await?; Ok(()) } @@ -5349,6 +4705,7 @@ mod tests { let test_provider = TestProvider::new(path, Arc::clone(&object_store)); let manifest_store = test_provider.manifest_store(); let table_store = test_provider.table_store(); + let wal_store = test_provider.wal_store(); let mut stored_manifest = StoredManifest::create_new_db( Arc::clone(&manifest_store), ManifestCore::new(), @@ -5404,9 +4761,12 @@ mod tests { oracle.clone(), None, ); + let status_manager = status_manager_for_core(&stored_manifest.manifest().core); + let wal_reader = Arc::new(native_wal_reader(&wal_store, &status_manager)); let inner = DbReaderInner { manifest_store, table_store, + wal_reader, options: DbReaderOptions { skip_wal_replay: true, ..DbReaderOptions::default() @@ -5417,10 +4777,9 @@ mod tests { oracle, reader, db_stats, - status_manager: DbStatusManager::new(0), + status_manager, segment_extractor: None, rand: test_provider.rand.clone(), - replay_tasks: ReplayTaskScope::new(), recorder, }; @@ -5470,6 +4829,8 @@ mod tests { ) -> DbReaderInner { let manifest_store = test_provider.manifest_store(); let table_store = test_provider.table_store(); + let wal_store = test_provider.wal_store(); + let status_manager = status_manager_for_core(current_core); let prior_state = ReaderState { generation: ReaderGeneration::new( @@ -5500,9 +4861,11 @@ mod tests { oracle.clone(), None, ); + let wal_reader = Arc::new(native_wal_reader(&wal_store, &status_manager)); DbReaderInner { manifest_store, table_store, + wal_reader, options: DbReaderOptions::default(), mode: DbReaderMode::ManagedCheckpoint, state: parking_lot::RwLock::new(Arc::new(prior_state)), @@ -5510,10 +4873,9 @@ mod tests { oracle, reader, db_stats, - status_manager: DbStatusManager::new(0), + status_manager, segment_extractor: None, rand: test_provider.rand.clone(), - replay_tasks: ReplayTaskScope::new(), recorder, } } @@ -5546,9 +4908,11 @@ mod tests { // RFC-0024: per-segment compactions, drains, and segment-set changes // are invisible to the root-tree diff. Verify the segments comparison // fires on each of those shapes. - use crate::db_state::{SortedRun, SsTableHandle, SsTableId, SsTableInfo, SsTableView}; - use crate::format::sst::SST_FORMAT_VERSION_LATEST; - use crate::manifest::{LsmTreeState, Segment}; + use crate::{ + db_state::{SortedRun, SsTableHandle, SsTableId, SsTableInfo, SsTableView}, + format::sst::SST_FORMAT_VERSION_LATEST, + manifest::{LsmTreeState, Segment}, + }; fn view(seq: u64) -> SsTableView { SsTableView::identity(SsTableHandle::new( @@ -5615,10 +4979,7 @@ mod tests { last_compacted_l0_sst_view_id: None, last_compacted_l0_sst_id: None, l0: VecDeque::from(vec![view(1)]), - compacted: vec![SortedRun { - id: 0, - sst_views: vec![view(4)], - }], + compacted: vec![SortedRun::new(0, [view(4)])], }, )]; assert!( @@ -5747,9 +5108,7 @@ mod tests { #[tokio::test] async fn should_record_incremental_wal_replay_metrics() { - use slatedb_common::metrics::{ - lookup_metric, lookup_metric_with_labels, DefaultMetricsRecorder, - }; + use slatedb_common::metrics::{lookup_metric, DefaultMetricsRecorder}; let object_store: Arc = Arc::new(InMemory::new()); let path = Path::from("/tmp/test_db_reader_wal_replay_metrics"); @@ -5769,7 +5128,7 @@ mod tests { &PutOptions::default(), &WriteOptions { await_durable: false, - ..WriteOptions::default() + ..Default::default() }, ) .await @@ -5804,42 +5163,6 @@ mod tests { Some(1), lookup_metric(&metrics_recorder, crate::db_stats::READER_REPLAY_MEMTABLES) ); - let list_metric = crate::db_stats::READER_WAL_REPLAY_OBJECT_STORE_CALLS; - assert!(lookup_metric_with_labels( - &metrics_recorder, - list_metric, - &[ - ( - crate::db_stats::REPLAY_SOURCE_LABEL, - crate::db_stats::REPLAY_SOURCE_READER_OPEN, - ), - ( - crate::db_stats::REPLAY_OPERATION_LABEL, - crate::db_stats::REPLAY_OPERATION_LIST, - ), - ], - ) - .is_some_and(|value| value > 0)); - for source in [ - crate::db_stats::REPLAY_SOURCE_RUNTIME_MANIFEST, - crate::db_stats::REPLAY_SOURCE_RUNTIME_TAIL, - ] { - assert_eq!( - lookup_metric_with_labels( - &metrics_recorder, - list_metric, - &[ - (crate::db_stats::REPLAY_SOURCE_LABEL, source), - ( - crate::db_stats::REPLAY_OPERATION_LABEL, - crate::db_stats::REPLAY_OPERATION_LIST, - ), - ], - ), - Some(0), - "runtime replay source unexpectedly performed LIST: {source}", - ); - } assert_eq!( Some(1), lookup_metric( @@ -5884,12 +5207,6 @@ mod tests { .build() .await .unwrap(); - let (tail_tx, tail_rx) = async_channel::unbounded(); - let mut pinned_poller = ManifestPoller::new(Arc::clone(&reader.inner), tail_tx, tail_rx); - assert!(!pinned_poller.runtime_tailing_enabled()); - assert_eq!(pinned_poller.tickers().len(), 1); - pinned_poller.start_runtime_tail(); - assert!(!pinned_poller.tail_in_flight); assert_eq!( Some(0), @@ -5966,6 +5283,16 @@ mod tests { )) } + fn wal_store(&self) -> Arc { + Arc::new(WalTableStore::new_with_fp_registry( + Arc::clone(&self.object_store), + SsTableFormat::default(), + PathResolver::from_root(self.path.clone()), + Arc::clone(&self.fp_registry), + TableStoreKind::Reader, + )) + } + fn manifest_store(&self) -> Arc { Arc::new(ManifestStore::new( &self.path, @@ -5983,8 +5310,7 @@ mod tests { let mut test_provider = TestProvider::new(path.clone(), Arc::clone(&object_store)); test_provider.system_clock = clock.clone(); - let merge_operator: crate::merge_operator::MergeOperatorType = - Arc::new(crate::test_utils::StringConcatMergeOperator); + let merge_operator: MergeOperatorType = Arc::new(test_utils::StringConcatMergeOperator); let db = Db::builder(path.clone(), Arc::clone(&object_store)) .with_settings(Settings { diff --git a/slatedb/src/db_state.rs b/slatedb/src/db_state.rs index f4070243c3..deebde0501 100644 --- a/slatedb/src/db_state.rs +++ b/slatedb/src/db_state.rs @@ -92,8 +92,8 @@ impl SsTableView { /// where no `DbRand` is available and the id is not stored in the manifest. pub(crate) fn identity(sst: SsTableHandle) -> Self { let id = match &sst.id { - SsTableId::Compacted(ulid) => *ulid, - SsTableId::Wal(wal_id) => Ulid::from_parts(*wal_id, 0), + Compacted(ulid) => *ulid, + Wal(wal_id) => Ulid::from_parts(*wal_id, 0), }; Self::new(id, sst) } @@ -385,7 +385,7 @@ impl SsTableId { } impl Debug for SsTableId { - fn fmt(&self, f: &mut std::fmt::Formatter) -> Result<(), std::fmt::Error> { + fn fmt(&self, f: &mut Formatter) -> Result<(), std::fmt::Error> { match self { Wal(id) => write!(f, "SsTableId::Wal({})", id), Compacted(id) => write!(f, "SsTableId::Compacted({})", id.to_string()), @@ -406,8 +406,8 @@ pub enum SstType { impl From<&SsTableId> for SstType { fn from(id: &SsTableId) -> Self { match id { - SsTableId::Wal(_) => SstType::Wal, - SsTableId::Compacted(_) => SstType::Compacted, + Wal(_) => SstType::Wal, + Compacted(_) => SstType::Compacted, } } } @@ -492,10 +492,27 @@ pub struct SortedRun { /// The unique identifier for this sorted run. pub id: u32, /// The list of SSTable views in this sorted run. - pub sst_views: Vec, + /// + /// Held behind an `Arc` so cloning a `SortedRun` (e.g. per read in the + /// scan path) is a single refcount bump rather than a deep clone of every + /// view's `Bytes` handles. + pub sst_views: Arc<[SsTableView]>, } impl SortedRun { + /// Create a sorted run from an ordered collection of SSTable views. + pub fn new(id: u32, sst_views: impl IntoIterator) -> Self { + Self { + id, + sst_views: sst_views.into_iter().collect(), + } + } + + /// Return the ordered SSTable views in this sorted run. + pub fn sst_views(&self) -> &[SsTableView] { + &self.sst_views + } + /// Estimate the total size of all SSTables in this sorted run. pub fn estimate_size(&self) -> u64 { self.sst_views.iter().map(|sst| sst.estimate_size()).sum() @@ -647,12 +664,12 @@ impl SortedRun { &self.sst_views[matching_range] } - pub(crate) fn into_tables_covering_range( - mut self, - range: &BytesRange, - ) -> VecDeque { + pub(crate) fn into_tables_covering_range(self, range: &BytesRange) -> VecDeque { let matching_range = self.table_idx_covering_range(range); - self.sst_views.drain(matching_range).collect() + // `sst_views` is shared behind an `Arc`, so we clone only the few + // covering views rather than draining the whole run. The full slice + // is released with a single refcount decrement when `self` drops. + self.sst_views[matching_range].iter().cloned().collect() } } @@ -848,8 +865,8 @@ mod tests { use proptest::proptest; use slatedb_common::clock::{DefaultSystemClock, SystemClock}; use std::collections::BTreeSet; - use std::collections::Bound::Included; use std::collections::VecDeque; + use std::ops::Bound::{Excluded, Included, Unbounded}; use std::ops::RangeBounds; use std::sync::Arc; @@ -1141,6 +1158,14 @@ mod tests { let sorted_first_keys: BTreeSet = table_first_keys.into_iter().collect(); let sorted_run = create_sorted_run(0, &sorted_first_keys); let covering_tables = sorted_run.tables_covering_range(range.clone()); + let borrowed_ids: Vec<_> = covering_tables.iter().map(|view| view.id).collect(); + let owned_ids: Vec<_> = sorted_run + .clone() + .into_tables_covering_range(&range) + .iter() + .map(|view| view.id) + .collect(); + assert_eq!(owned_ids, borrowed_ids); let first_key = sorted_first_keys.first().unwrap().clone(); let range_start_key = test_utils::bound_as_option(range.start_bound()) @@ -1175,15 +1200,15 @@ mod tests { #[test] fn test_sorted_run_collect_tables_for_point_key() { - let sorted_run = SortedRun { - id: 0, - sst_views: vec![ + let sorted_run = SortedRun::new( + 0, + [ create_compacted_sst_view_with_bounds(b"a", Some(b"k")), create_compacted_sst_view_with_bounds(b"k", Some(b"k")), create_compacted_sst_view_with_bounds(b"k", Some(b"m")), create_compacted_sst_view_with_bounds(b"z", Some(b"z")), ], - }; + ); let covering_tables = sorted_run.tables_covering_point_key(b"k"); assert_eq!(covering_tables.len(), 3); @@ -1203,15 +1228,21 @@ mod tests { assert!(sorted_run.tables_covering_point_key(b"0").is_empty()); } + #[test] + fn test_sorted_run_clone_shares_sst_views() { + let sorted_run = SortedRun::new(0, [create_compacted_sst_view(Some(Bytes::from("a")))]); + let cloned = sorted_run.clone(); + + assert!(Arc::ptr_eq(&sorted_run.sst_views, &cloned.sst_views)); + assert_eq!(sorted_run.sst_views(), cloned.sst_views()); + } + fn create_sorted_run(id: u32, first_keys: &BTreeSet) -> SortedRun { let mut ssts = Vec::new(); for first_key in first_keys { ssts.push(create_compacted_sst_view(Some(first_key.clone()))); } - SortedRun { - id, - sst_views: ssts, - } + SortedRun::new(id, ssts) } fn create_compacted_sst_view(first_entry: Option) -> SsTableView { @@ -1244,7 +1275,7 @@ mod tests { #[test] fn max_l0_overlap_empty_is_zero() { - let l0: std::collections::VecDeque = std::collections::VecDeque::new(); + let l0: VecDeque = VecDeque::new(); assert_eq!(super::max_l0_overlap(&l0), 0); } @@ -1252,7 +1283,7 @@ mod tests { fn max_l0_overlap_disjoint_ranges_is_one() { // Simulates a post-union manifest where each source's L0s cover // disjoint key ranges — the peak per-point count stays at 1. - let mut l0 = std::collections::VecDeque::new(); + let mut l0 = VecDeque::new(); l0.push_back(create_compacted_sst_view_with_bounds(b"a", Some(b"b"))); l0.push_back(create_compacted_sst_view_with_bounds(b"c", Some(b"d"))); l0.push_back(create_compacted_sst_view_with_bounds(b"e", Some(b"f"))); @@ -1262,7 +1293,7 @@ mod tests { #[test] fn max_l0_overlap_full_overlap_counts_all() { - let mut l0 = std::collections::VecDeque::new(); + let mut l0 = VecDeque::new(); for _ in 0..4 { l0.push_back(create_compacted_sst_view_with_bounds(b"a", Some(b"z"))); } @@ -1272,7 +1303,7 @@ mod tests { #[test] fn max_l0_overlap_partial_overlap() { // A: [a, c], B: [b, d]. At B.start=b, both A and B contain b. - let mut l0 = std::collections::VecDeque::new(); + let mut l0 = VecDeque::new(); l0.push_back(create_compacted_sst_view_with_bounds(b"a", Some(b"c"))); l0.push_back(create_compacted_sst_view_with_bounds(b"b", Some(b"d"))); assert_eq!(super::max_l0_overlap(&l0), 2); @@ -1281,7 +1312,7 @@ mod tests { #[test] fn max_l0_overlap_mixed_disjoint_groups() { // Two disjoint groups of 3 overlapping SSTs each. Peak is 3, not 6. - let mut l0 = std::collections::VecDeque::new(); + let mut l0 = VecDeque::new(); for _ in 0..3 { l0.push_back(create_compacted_sst_view_with_bounds(b"a", Some(b"c"))); } @@ -1294,7 +1325,7 @@ mod tests { #[test] fn max_l0_overlap_single_point_range_is_one() { // A view whose first_entry == last_entry covers exactly one key. - let mut l0 = std::collections::VecDeque::new(); + let mut l0 = VecDeque::new(); l0.push_back(create_compacted_sst_view_with_bounds(b"k", Some(b"k"))); assert_eq!(super::max_l0_overlap(&l0), 1); } @@ -1302,7 +1333,7 @@ mod tests { #[test] fn max_l0_overlap_many_point_ranges_same_key() { // N coincident point ranges [k, k] all cover key k → peak N. - let mut l0 = std::collections::VecDeque::new(); + let mut l0 = VecDeque::new(); for _ in 0..5 { l0.push_back(create_compacted_sst_view_with_bounds(b"k", Some(b"k"))); } @@ -1313,7 +1344,7 @@ mod tests { fn max_l0_overlap_mixed_point_and_longer_ranges_at_same_key() { // Two point ranges [k, k] and two longer ranges [k, z] all cover k. // Peak at k is 4; past k, only the two longer ranges remain. - let mut l0 = std::collections::VecDeque::new(); + let mut l0 = VecDeque::new(); l0.push_back(create_compacted_sst_view_with_bounds(b"k", Some(b"k"))); l0.push_back(create_compacted_sst_view_with_bounds(b"k", Some(b"k"))); l0.push_back(create_compacted_sst_view_with_bounds(b"k", Some(b"z"))); @@ -1324,7 +1355,7 @@ mod tests { #[test] fn max_l0_overlap_edge_touching_inclusive_counts_both() { // [a, b] and [b, c]: both contain b → peak 2. - let mut l0 = std::collections::VecDeque::new(); + let mut l0 = VecDeque::new(); l0.push_back(create_compacted_sst_view_with_bounds(b"a", Some(b"b"))); l0.push_back(create_compacted_sst_view_with_bounds(b"b", Some(b"c"))); assert_eq!(super::max_l0_overlap(&l0), 2); @@ -1336,14 +1367,10 @@ mod tests { // First view has an Excluded end at b via a visible_range projection. let a = Bytes::copy_from_slice(b"a"); let b = Bytes::copy_from_slice(b"b"); - let v1 = create_compacted_sst_view_with_bounds(b"a", Some(b"b")).with_visible_range( - BytesRange::new( - std::ops::Bound::Included(a), - std::ops::Bound::Excluded(b.clone()), - ), - ); + let v1 = create_compacted_sst_view_with_bounds(b"a", Some(b"b")) + .with_visible_range(BytesRange::new(Included(a), Excluded(b.clone()))); let v2 = create_compacted_sst_view_with_bounds(b"b", Some(b"c")); - let mut l0 = std::collections::VecDeque::new(); + let mut l0 = VecDeque::new(); l0.push_back(v1); l0.push_back(v2); assert_eq!(super::max_l0_overlap(&l0), 1); @@ -1353,7 +1380,7 @@ mod tests { fn max_l0_overlap_unbounded_end_single_view() { // A view with first_entry but no last_entry has effective_range // [first, Unbounded) — still one view, peak 1. - let mut l0 = std::collections::VecDeque::new(); + let mut l0 = VecDeque::new(); l0.push_back(create_compacted_sst_view_with_bounds(b"a", None)); assert_eq!(super::max_l0_overlap(&l0), 1); } @@ -1362,7 +1389,7 @@ mod tests { fn max_l0_overlap_unbounded_ends_share_tail() { // [a, ∞) and [b, ∞) both extend to +∞, so they overlap at every // point ≥ b. Peak is 2. - let mut l0 = std::collections::VecDeque::new(); + let mut l0 = VecDeque::new(); l0.push_back(create_compacted_sst_view_with_bounds(b"a", None)); l0.push_back(create_compacted_sst_view_with_bounds(b"b", None)); assert_eq!(super::max_l0_overlap(&l0), 2); @@ -1372,7 +1399,7 @@ mod tests { fn max_l0_overlap_mixed_bounded_and_unbounded_end() { // [a, m] ends at m; [b, ∞) starts before m and extends past it. // They coexist on [b, m]. Peak is 2. - let mut l0 = std::collections::VecDeque::new(); + let mut l0 = VecDeque::new(); l0.push_back(create_compacted_sst_view_with_bounds(b"a", Some(b"m"))); l0.push_back(create_compacted_sst_view_with_bounds(b"b", None)); assert_eq!(super::max_l0_overlap(&l0), 2); @@ -1384,11 +1411,10 @@ mod tests { // Effective range becomes [m, z] (physical end clamps the Unbounded). // Pair with [n, ∞): overlap on [n, z]. Peak is 2. let m = Bytes::copy_from_slice(b"m"); - let projected = create_compacted_sst_view_with_bounds(b"a", Some(b"z")).with_visible_range( - BytesRange::new(std::ops::Bound::Included(m), std::ops::Bound::Unbounded), - ); + let projected = create_compacted_sst_view_with_bounds(b"a", Some(b"z")) + .with_visible_range(BytesRange::new(Included(m), Unbounded)); let open = create_compacted_sst_view_with_bounds(b"n", None); - let mut l0 = std::collections::VecDeque::new(); + let mut l0 = VecDeque::new(); l0.push_back(projected); l0.push_back(open); assert_eq!(super::max_l0_overlap(&l0), 2); @@ -1401,19 +1427,11 @@ mod tests { let lo = Bytes::copy_from_slice(b"a"); let mid = Bytes::copy_from_slice(b"m"); let hi = Bytes::copy_from_slice(b"z"); - let v1 = create_compacted_sst_view_with_bounds(b"a", Some(b"z")).with_visible_range( - BytesRange::new( - std::ops::Bound::Included(lo.clone()), - std::ops::Bound::Excluded(mid.clone()), - ), - ); - let v2 = create_compacted_sst_view_with_bounds(b"a", Some(b"z")).with_visible_range( - BytesRange::new( - std::ops::Bound::Included(mid), - std::ops::Bound::Included(hi), - ), - ); - let mut l0 = std::collections::VecDeque::new(); + let v1 = create_compacted_sst_view_with_bounds(b"a", Some(b"z")) + .with_visible_range(BytesRange::new(Included(lo.clone()), Excluded(mid.clone()))); + let v2 = create_compacted_sst_view_with_bounds(b"a", Some(b"z")) + .with_visible_range(BytesRange::new(Included(mid), Included(hi))); + let mut l0 = VecDeque::new(); l0.push_back(v1); l0.push_back(v2); assert_eq!(super::max_l0_overlap(&l0), 1); @@ -1453,7 +1471,7 @@ mod tests { }); proptest!(ProptestConfig::with_cases(256), |(specs in vec(spec, 0..=8))| { - let mut l0 = std::collections::VecDeque::new(); + let mut l0 = VecDeque::new(); for s in &specs { let view = match &s.end { EndKind::Inclusive(end) => { @@ -1466,8 +1484,8 @@ mod tests { &s.start, None, ) .with_visible_range(BytesRange::new( - std::ops::Bound::Included(s.start.clone()), - std::ops::Bound::Excluded(end.clone()), + Included(s.start.clone()), + Excluded(end.clone()), )), }; l0.push_back(view); diff --git a/slatedb/src/db_stats.rs b/slatedb/src/db_stats.rs index cb8c8fe5c8..7860d95b7e 100644 --- a/slatedb/src/db_stats.rs +++ b/slatedb/src/db_stats.rs @@ -105,7 +105,9 @@ pub(crate) struct DbStatsInner { pub(crate) reader_replay_memtables: Arc, pub(crate) reader_active_checkpoints: Arc, pub(crate) reader_manifest_polls: Arc, + #[allow(dead_code)] pub(crate) reader_wal_replay_list_reader_open: Arc, + #[allow(dead_code)] pub(crate) reader_wal_replay_list_checkpoint_recovery: Arc, } diff --git a/slatedb/src/db_status.rs b/slatedb/src/db_status.rs index 2448db8ce6..e9b7b6f7ba 100644 --- a/slatedb/src/db_status.rs +++ b/slatedb/src/db_status.rs @@ -1,11 +1,14 @@ use std::collections::BTreeSet; +use std::fmt; +use std::sync::Arc; use bytes::Bytes; +use futures::future::BoxFuture; use tokio::sync::watch; use crate::error::SlateDBError; use crate::manifest::VersionedManifest; -use crate::utils::WatchableOnceCell; +use crate::utils::{WatchableOnceCell, WatchableOnceCellReader}; use crate::CloseReason; /// A segment (RFC-0024), identified by the key prefix it owns; the segment @@ -64,17 +67,27 @@ impl DbStatus { } } -pub(crate) trait ClosedResultWriter: std::fmt::Debug + Send + Sync + 'static { +pub(crate) trait ClosedResultWriter: fmt::Debug + Send + Sync + 'static { fn write_result(&self, result: Result<(), SlateDBError>); - fn result_reader(&self) -> crate::utils::WatchableOnceCellReader>; + fn result_reader(&self) -> WatchableOnceCellReader>; } /// Manages database lifecycle status, including the close result and /// status subscriptions. -#[derive(Clone, Debug)] +#[derive(Clone)] pub(crate) struct DbStatusManager { cell: WatchableOnceCell>, tx: watch::Sender, + durability_waiter: DurabilityWaiter, +} + +impl fmt::Debug for DbStatusManager { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.debug_struct("DbStatusManager") + .field("cell", &self.cell) + .field("tx", &self.tx) + .finish_non_exhaustive() + } } impl DbStatusManager { @@ -103,9 +116,13 @@ impl DbStatusManager { memtable_segments: initial_memtable_segments, close_reason: None, }); + let cell = WatchableOnceCell::new(); + let durability_waiter = new_durability_waiter(tx.subscribe(), cell.reader()); + Self { - cell: WatchableOnceCell::new(), + cell, tx, + durability_waiter, } } @@ -214,6 +231,11 @@ impl DbStatusManager { self.tx.subscribe() } + /// Returns the shared [`DurabilityWaiter`] for this database. + pub(crate) fn durability_waiter(&self) -> DurabilityWaiter { + Arc::clone(&self.durability_waiter) + } + pub(crate) fn status(&self) -> DbStatus { self.tx.borrow().clone() } @@ -224,7 +246,7 @@ impl ClosedResultWriter for WatchableOnceCell> { self.write(result); } - fn result_reader(&self) -> crate::utils::WatchableOnceCellReader> { + fn result_reader(&self) -> WatchableOnceCellReader> { self.reader() } } @@ -240,11 +262,50 @@ impl ClosedResultWriter for DbStatusManager { } } - fn result_reader(&self) -> crate::utils::WatchableOnceCellReader> { + fn result_reader(&self) -> WatchableOnceCellReader> { self.cell.reader() } } +/// Shared callback used by [`crate::WriteHandle`]s to wait for a sequence +/// number to become durable. +pub(crate) type DurabilityWaiter = + Arc BoxFuture<'static, Result<(), crate::Error>> + Send + Sync + 'static>; + +/// Creates a [`DurabilityWaiter`] backed by database status and close-result +/// readers. +/// +/// The waiter captures only readers, so cloning it into a write handle does not +/// keep the database's status sender alive after the database is dropped. +fn new_durability_waiter( + status_rx: watch::Receiver, + close_result: WatchableOnceCellReader>, +) -> DurabilityWaiter { + Arc::new(move |seq| -> BoxFuture<'static, Result<(), crate::Error>> { + let mut status_rx = status_rx.clone(); + let close_result = close_result.clone(); + + Box::pin(async move { + let wait_result = status_rx + .wait_for(|status| status.durable_seq >= seq || status.close_reason.is_some()) + .await; + + match wait_result { + Ok(status) if status.durable_seq >= seq => Ok(()), + // The write was not durable before the database closed. Use the + // recorded close result to preserve fencing and panic errors. + Ok(_) | Err(_) => match close_result + .read() + .expect("database closed without recording a close result") + { + Ok(()) => Err(SlateDBError::Closed.into()), + Err(error) => Err(error.into()), + }, + } + }) + }) +} + #[cfg(test)] mod tests { use super::*; diff --git a/slatedb/src/db_transaction.rs b/slatedb/src/db_transaction.rs index d1e832745c..70622121cc 100644 --- a/slatedb/src/db_transaction.rs +++ b/slatedb/src/db_transaction.rs @@ -48,6 +48,11 @@ impl DisjointMergeBatchEntry { /// configurable isolation levels. This is the main interface for transactional /// operations in SlateDB. /// +/// Committing a non-empty transaction returns a [`WriteHandle`] without +/// waiting for durability. Call [`WriteHandle::await_durable`] on that handle +/// to wait for that commit, or call [`crate::Db::flush`] to flush all pending +/// writes. +/// /// # Examples /// /// Basic transaction usage: @@ -260,7 +265,7 @@ impl DbTransaction { let db_state = self.db_inner.state.read().view(); - let mut key_to_idx = std::collections::HashMap::::with_capacity(keys.len()); + let mut key_to_idx = HashMap::::with_capacity(keys.len()); let mut unique_keys = Vec::::with_capacity(keys.len()); let mut output_positions = Vec::>::with_capacity(keys.len()); @@ -1101,10 +1106,14 @@ impl DbTransaction { /// Commit the transaction by applying all buffered operations to the database. /// - /// This method finalizes the transaction by writing all pending puts, deletes, and other - /// operations from the write batch to persistent storage. The actual conflict detection - /// (including read-write and write-write conflicts) is deferred to the task that processes - /// the WriteBatch, which ensures the atomicity of transactions. + /// This method finalizes the transaction by writing all pending puts, + /// deletes, and other operations from the write batch to the in-memory WAL + /// and MemTable. The actual conflict detection (including read-write and + /// write-write conflicts) is deferred to the task that processes the + /// WriteBatch, which ensures the atomicity of transactions. + /// + /// The default write options wait for a successful commit to become + /// durable in object storage. /// /// If the transaction's write batch is empty, this operation is a no-op and returns `Ok(())` /// immediately without any database interaction. Since it's impossible to have read-write @@ -1125,7 +1134,12 @@ impl DbTransaction { /// Commit the transaction with custom write options. /// /// This method behaves the same as [`DbTransaction::commit`], but allows callers - /// to specify custom [`WriteOptions`], such as `await_durable`. + /// to specify custom [`WriteOptions`]. + /// + /// Durability behavior is controlled by [`WriteOptions::await_durable`]. + /// When it is `false`, call [`WriteHandle::await_durable`] on the returned + /// handle when the result is `Some`, or call [`crate::Db::flush`] to flush + /// all pending writes. /// /// ## Arguments /// - `options`: the write options to use for the commit @@ -1411,6 +1425,7 @@ mod tests { use crate::merge_operator::{MergeOperator, MergeOperatorError}; use crate::object_store::memory::InMemory; use rstest::rstest; + use std::mem::size_of; use std::sync::Arc; struct CounterMergeOperator; @@ -2068,14 +2083,14 @@ mod tests { } #[tokio::test(flavor = "multi_thread", worker_threads = 4)] - async fn test_txn_commit_await_durable_false() { + async fn test_txn_commit_returns_before_durable() { use crate::config::{DurabilityLevel::*, ReadOptions, WriteOptions}; use fail_parallel::FailPointRegistry; // Setup database with failpoints to pause durable writes let fp_registry = Arc::new(FailPointRegistry::new()); let object_store: Arc = Arc::new(InMemory::new()); - let db = crate::Db::builder("/tmp/test_txn_commit_await_durable_false", object_store) + let db = crate::Db::builder("/tmp/test_txn_commit_returns_before_durable", object_store) .with_fp_registry(fp_registry.clone()) .build() .await @@ -2088,7 +2103,7 @@ mod tests { let txn = db.begin(IsolationLevel::Snapshot).await.unwrap(); txn.put(b"k", b"v").unwrap(); - // Commit without waiting for durability + // Commits return without waiting for durability. txn.commit_with_options(&WriteOptions { await_durable: false, ..Default::default() @@ -3223,7 +3238,7 @@ mod tests { let object_store: Arc = Arc::new(InMemory::new()); let clock = Arc::new(MockSystemClock::new()); let settings = crate::config::Settings { - default_ttl: Some(10), + default_ttl_millis: Some(10), ..Default::default() }; let db = crate::Db::builder("disjoint_no_default_ttl", object_store) @@ -3272,8 +3287,8 @@ mod tests { let object_store: Arc = Arc::new(InMemory::new()); let metrics = Arc::new(DefaultMetricsRecorder::new()); let settings = crate::config::Settings { - max_transaction_conflict_metadata_bytes: core::mem::size_of::(), - max_retained_conflict_metadata_bytes: 2 * core::mem::size_of::(), + max_transaction_conflict_metadata_bytes: size_of::(), + max_retained_conflict_metadata_bytes: 2 * size_of::(), ..Default::default() }; let db = crate::Db::builder("disjoint_metadata_limits", object_store) @@ -3294,7 +3309,7 @@ mod tests { txn.commit().await.unwrap(); assert_eq!( lookup_metric(&metrics, crate::db_stats::CONFLICT_METADATA_RETAINED_BYTES), - Some(((ordinal + 1) * core::mem::size_of::()) as i64) + Some(((ordinal + 1) * size_of::()) as i64) ); assert_eq!( lookup_metric(&metrics, crate::db_stats::CONFLICT_METADATA_RETAINED_TOKENS), @@ -3732,7 +3747,7 @@ mod tests { b"counter", 1u64.to_le_bytes(), &MergeOptions { - ttl: crate::config::Ttl::ExpireAfter(3600), + ttl: crate::config::Ttl::ExpireAfterMillis(3600), }, ) .unwrap(); @@ -3740,7 +3755,7 @@ mod tests { b"counter", 2u64.to_le_bytes(), &MergeOptions { - ttl: crate::config::Ttl::ExpireAfter(7200), + ttl: crate::config::Ttl::ExpireAfterMillis(7200), }, ) .unwrap(); @@ -3781,7 +3796,7 @@ mod tests { object_store_cache_options: crate::config::ObjectStoreCacheOptions::default(), garbage_collector_options: None, metric_level: MetricLevel::default(), - default_ttl: None, + default_ttl_millis: None, max_transaction_conflict_metadata_bytes: crate::config::DEFAULT_MAX_TRANSACTION_CONFLICT_METADATA_BYTES, max_retained_conflict_metadata_bytes: @@ -3824,7 +3839,7 @@ mod tests { clock.set(200); let txn = db.begin(IsolationLevel::Snapshot).await.unwrap(); let put_opts = PutOptions { - ttl: crate::config::Ttl::ExpireAfter(1000), + ttl: crate::config::Ttl::ExpireAfterMillis(1000), }; txn.put_with_options(b"key2", b"value2", &put_opts).unwrap(); let handle = txn diff --git a/slatedb/src/dispatcher.rs b/slatedb/src/dispatcher.rs index 0dc7aac67d..c6dad6d6aa 100644 --- a/slatedb/src/dispatcher.rs +++ b/slatedb/src/dispatcher.rs @@ -983,7 +983,7 @@ mod test { async fn cleanup( &mut self, - mut messages: futures::stream::BoxStream<'async_trait, TestMessage>, + mut messages: BoxStream<'async_trait, TestMessage>, result: Result<(), SlateDBError>, ) -> Result<(), SlateDBError> { self.cleanup_called.write(result); diff --git a/slatedb/src/error.rs b/slatedb/src/error.rs index b10cdfa5e3..2547a18ed1 100644 --- a/slatedb/src/error.rs +++ b/slatedb/src/error.rs @@ -8,7 +8,7 @@ use uuid::Uuid; use crate::bytes_range::BytesRange; use crate::error::SlateDBError::{ - LatestTransactionalObjectVersionMissing, TransactionalObjectVersionExists, + DatabaseMissing, LatestTransactionalObjectVersionMissing, TransactionalObjectVersionExists, }; use crate::merge_operator::MergeOperatorError; use slatedb_txn_obj::TransactionalObjectError; @@ -101,8 +101,8 @@ pub(crate) enum SlateDBError { #[error("wal store reconfiguration unsupported")] WalStoreReconfigurationError, - #[error("wal truncated")] - WalTruncated, + #[error("wal truncated at wal file `{0}`")] + WalTruncated(u64), #[error("wal unavailable")] WalUnavailable(Arc), @@ -251,8 +251,8 @@ pub(crate) enum SlateDBError { #[error("clone source paths must be unique, found duplicate: `{0}`")] DuplicatedCloneSourcePath(Path), - #[error("Manifest union of sources with WAL is not supported, source with WAL: `{paths:?}`")] - InvalidUnionSourceWithWal { paths: Vec }, + #[error("Projection and/or union with WAL is not supported, sources with WAL: `{paths:?}`")] + InvalidCloneSourceWithWal { paths: Vec }, #[error("Source manifest set must not be empty")] InvalidUnionSetEmpty(), @@ -270,6 +270,7 @@ pub(crate) enum SlateDBError { InvalidManifestPollInterval(Duration), #[error("invalid WAL poll interval. interval=`{0:?}`")] + #[allow(dead_code)] InvalidWalPollInterval(Duration), #[error("checkpoint lifetime must be at least double the manifest poll interval. lifetime=`{lifetime:?}`, interval=`{interval:?}`")] @@ -278,6 +279,9 @@ pub(crate) enum SlateDBError { interval: Duration, }, + #[error("invalid sst batch size. size=`{0}`")] + InvalidSSTBatchSize(usize), + #[error("invalid configuration: {0}")] InvalidConfiguration(String), @@ -739,6 +743,7 @@ impl From for Error { } SlateDBError::InvalidObjectStorePath(_) => Error::invalid(msg), SlateDBError::UnknownConfigurationFormat(_) => Error::invalid(msg), + SlateDBError::InvalidSSTBatchSize(_) => Error::invalid(msg), SlateDBError::InvalidConfiguration(_) => Error::invalid(msg), SlateDBError::InvalidCheckpointLifetime(_) => Error::invalid(msg), SlateDBError::InvalidManifestPollInterval(_) @@ -749,7 +754,7 @@ impl From for Error { SlateDBError::SeekKeyLessThanLastReturnedKey => Error::invalid(msg), SlateDBError::IdenticalClonePaths { .. } => Error::invalid(msg), SlateDBError::DuplicatedCloneSourcePath(_) => Error::invalid(msg), - SlateDBError::InvalidUnionSourceWithWal { .. } => Error::invalid(msg), + SlateDBError::InvalidCloneSourceWithWal { .. } => Error::invalid(msg), SlateDBError::InvalidUnionSetEmpty() => Error::invalid(msg), SlateDBError::InvalidUnion(_) => Error::invalid(msg), SlateDBError::InvalidProjection { .. } => Error::invalid(msg), @@ -793,9 +798,9 @@ impl From for Error { SlateDBError::CheckpointAlreadyExists(_) => Error::data(msg), SlateDBError::InvalidVersion { .. } => Error::data(msg), SlateDBError::ManifestMissing(_) => Error::data(msg), - SlateDBError::LatestTransactionalObjectVersionMissing => Error::data(msg), - SlateDBError::DatabaseMissing => Error::data(msg).with_code(ErrorCode::DatabaseMissing), - SlateDBError::TransactionalObjectVersionExists => Error::data(msg), + LatestTransactionalObjectVersionMissing => Error::data(msg), + DatabaseMissing => Error::data(msg).with_code(ErrorCode::DatabaseMissing), + TransactionalObjectVersionExists => Error::data(msg), SlateDBError::InvalidTransactionalObjectState => Error::data(msg), SlateDBError::EmptyManifest => Error::data(msg), SlateDBError::EmptyBlock => Error::data(msg), @@ -809,6 +814,7 @@ impl From for Error { SlateDBError::CloneExternalDbMissing => Error::data(msg), SlateDBError::CloneIncorrectExternalDbCheckpoint { .. } => Error::data(msg), SlateDBError::CloneIncorrectFinalCheckpoint { .. } => Error::data(msg), + SlateDBError::WalTruncated(_) => Error::data(msg), SlateDBError::WalDataError(src) => Error::data(msg).with_source(Box::new(src)), // Internal errors @@ -824,7 +830,6 @@ impl From for Error { SlateDBError::TransactionalObjectError(err) => { Error::internal(msg).with_source(Box::new(err)) } - SlateDBError::WalTruncated => Error::internal(msg), SlateDBError::WalInternalError(src) => Error::internal(msg).with_source(Box::new(src)), } } @@ -868,7 +873,7 @@ mod tests { #[test] fn database_missing_has_stable_code_without_changing_broad_kind() { - let public_err = Error::from(SlateDBError::DatabaseMissing); + let public_err = Error::from(DatabaseMissing); assert_eq!(public_err.kind(), ErrorKind::Data); assert_eq!(public_err.code(), Some(ErrorCode::DatabaseMissing)); @@ -877,7 +882,7 @@ mod tests { #[test] fn other_data_errors_do_not_claim_database_is_missing() { for err in [ - SlateDBError::LatestTransactionalObjectVersionMissing, + LatestTransactionalObjectVersionMissing, SlateDBError::ManifestMissing(7), SlateDBError::InvalidDBState, ] { diff --git a/slatedb/src/fence.rs b/slatedb/src/fence.rs index 734c02e79a..863fafa512 100644 --- a/slatedb/src/fence.rs +++ b/slatedb/src/fence.rs @@ -1,35 +1,34 @@ use crate::dispatcher::MessageHandlerExecutor; use crate::error::SlateDBError; use crate::manifest::store::{FenceableManifest, StoredManifest}; -use crate::tablestore::TableStore; use crate::utils::WatchableOnceCellReader; -use crate::wal::writer_init::{WalWriterInit, WalWriterInitOptions}; -use crate::wal::{WalWriter, WriterInit}; +use crate::wal::slatedb::store::WalTableStore; +use crate::wal::slatedb::writer_init::{SlateDbWalWriterInit, SlateDbWalWriterInitOptions}; +use crate::wal::{WalIterator, WalWriter, WriterInit}; use crate::Settings; use fail_parallel::{fail_point_send, FailPointTx}; -use log::error; use slatedb_common::metrics::MetricsRecorderHelper; use slatedb_common::SystemClock; use std::num::NonZeroU64; -use std::ops::Range; use std::sync::Arc; use std::time::Duration; pub(crate) struct WriterFencer { closed_result_reader: WatchableOnceCellReader>, recorder: MetricsRecorderHelper, - wal_writer_init_options: WalWriterInitOptions, - table_store: Arc, + wal_writer_init_options: SlateDbWalWriterInitOptions, + wal_store: Arc, manifest_update_timeout: Duration, system_clock: Arc, task_executor: Arc, + wal_writer_init: Option>, #[cfg_attr(not(test), allow(dead_code))] fp_tx: FailPointTx, } pub(crate) struct WriterFenceResult { pub(crate) manifest: FenceableManifest, - pub(crate) replay_range: Range, + pub(crate) replay_iterator: Box, pub(crate) wal_writer: Box, } @@ -37,18 +36,20 @@ impl WriterFencer { pub(crate) fn new( closed_result_reader: WatchableOnceCellReader>, recorder: MetricsRecorderHelper, - table_store: Arc, + wal_store: Arc, settings: &Settings, system_clock: Arc, task_executor: Arc, + wal_writer_init: Option>, ) -> Self { Self::new_with_fp_handle( closed_result_reader, recorder, - table_store, + wal_store, settings, system_clock, task_executor, + wal_writer_init, FailPointTx::dummy(), ) } @@ -56,20 +57,22 @@ impl WriterFencer { fn new_with_fp_handle( closed_result_reader: WatchableOnceCellReader>, recorder: MetricsRecorderHelper, - table_store: Arc, + wal_store: Arc, settings: &Settings, system_clock: Arc, task_executor: Arc, + wal_writer_init: Option>, fp_tx: FailPointTx, ) -> Self { Self { closed_result_reader, recorder, - table_store, + wal_store, wal_writer_init_options: settings.into(), manifest_update_timeout: settings.manifest_update_timeout, system_clock, task_executor, + wal_writer_init, fp_tx, } } @@ -79,25 +82,29 @@ impl WriterFencer { } /// Fences all writers with an older epoch than the provided `stored_manifest` by (1) writing - /// a new `FenceableManifest` with a bumped epoch, and (2) writing an empty WAL file that acts - /// as a barrier. Any parallel old writers will fail with `SlateDBError::Fenced` when trying - /// to "re-write" this file. Returns a `WriterFence` with the `FenceableManifest` and range - /// that must be replayed to recover up to the current epoch + /// a new `FenceableManifest` with a bumped epoch, and (2) delegating WAL fencing and recovery + /// to the configured [`WriterInit`]. Returns the fenced manifest, the initialized WAL writer, + /// and the iterator that must be replayed to recover up to the current epoch. pub(crate) async fn fence( - self, + mut self, stored_manifest: StoredManifest, writer_epoch: Option, ) -> Result { - let wal_writer_init = WalWriterInit::load( - self.closed_result_reader.clone(), - self.recorder.clone(), - self.table_store.clone(), - self.wal_writer_init_options, - stored_manifest.manifest(), - self.task_executor.clone(), - self.fp_tx.clone(), - ) - .await?; + let wal_writer_init = match self.wal_writer_init.take() { + Some(wal_writer_init) => wal_writer_init, + None => Box::new( + SlateDbWalWriterInit::load( + self.closed_result_reader.clone(), + self.recorder.clone(), + self.wal_store.clone(), + self.wal_writer_init_options, + stored_manifest.manifest(), + self.task_executor.clone(), + self.fp_tx.clone(), + ) + .await?, + ), + }; let manifest = match writer_epoch { Some(writer_epoch) => { @@ -128,17 +135,10 @@ impl WriterFencer { manifest.refresh().await?; fail_point_send!(self.fp_tx, "FinalRefreshManifest"); - let replay_range = match result.replay_range.try_into() { - Ok(replay_range) => replay_range, - Err(_) => { - error!("replay range must use inclusive lower bound and exclusive upper bound"); - return Err(SlateDBError::InvalidDBState); - } - }; Ok(WriterFenceResult { manifest, wal_writer: result.wal_writer, - replay_range, + replay_iterator: result.replay_iterator, }) } } @@ -159,8 +159,10 @@ mod tests { use crate::manifest::ManifestCore; use crate::memtable_flusher::MANIFEST_REFRESH_COUNT; use crate::object_stores::ObjectStores; + use crate::paths::PathResolver; use crate::tablestore::{TableStore, TableStoreKind}; use crate::utils::WatchableOnceCell; + use crate::wal::slatedb::store::WalTableStore; use crate::{CloseReason, Db, ErrorKind, Settings}; use bytes::Bytes; use fail_parallel::fail_point_channel; @@ -194,6 +196,7 @@ mod tests { let object_store: Arc = Arc::new(InMemory::new()); let settings = test_db_options(); let system_clock: Arc = Arc::new(DefaultSystemClock::new()); + let fp_registry = Arc::new(FailPointRegistry::new()); let manifest_store = Arc::new(ManifestStore::new(&Path::from(path), object_store.clone())); let table_store = Arc::new(TableStore::new( @@ -204,6 +207,13 @@ mod tests { TableStoreKind::Main, BlockCachePolicy::default(), )); + let wal_store = Arc::new(WalTableStore::new_with_fp_registry( + Arc::clone(&object_store), + SsTableFormat::default(), + PathResolver::from_root(path), + Arc::clone(&fp_registry), + TableStoreKind::Main, + )); let stored_manifest = StoredManifest::create_new_db( manifest_store.clone(), ManifestCore::new(), @@ -211,7 +221,6 @@ mod tests { ) .await .unwrap(); - let fp_registry = Arc::new(FailPointRegistry::new()); let (fp_tx, event_rx) = fail_point_channel(fp_registry.clone()); let cell = Arc::new(WatchableOnceCell::new()); let recorder = MetricsRecorderHelper::new( @@ -225,10 +234,11 @@ mod tests { let fencer = WriterFencer::new_with_fp_handle( cell.reader(), recorder, - table_store.clone(), + wal_store, &settings, system_clock.clone(), task_executor.clone(), + None, fp_tx, ); Self { @@ -307,6 +317,7 @@ mod tests { &MetricsRecorderHelper::noop(), Arc::new(DefaultSystemClock::new()), None, + None, ); gc.run_gc_once().await; // verify all regular (size > 0) wals up to wal_id are deleted (the wal @@ -330,7 +341,10 @@ mod tests { async fn put(&mut self, db: &Db, v: u32, expect_fenced: bool) { let k = Bytes::from(format!("k{}", v)); let v = Bytes::from(format!("v{}", v)); - let result = db.put(k.as_ref(), v.as_ref()).await; + let result = match db.put(k.as_ref(), v.as_ref()).await { + Ok(handle) => handle.await_durable().await, + Err(error) => Err(error), + }; if expect_fenced { assert_eq!( result.unwrap_err().kind(), @@ -601,14 +615,26 @@ mod tests { // unpause WriterFencer fail_parallel::cfg(h.fp_registry.clone(), case.event, "off").unwrap(); // verify it returns successfully - let result = jh.await.unwrap().unwrap(); + let mut result = jh.await.unwrap().unwrap(); // The fencer's stale empty_wal_id was retried above the fenced writer's possibly // advanced replay_after_wal_id. - assert_eq!(result.replay_range.start, replay_after_wal_id + 1); + let first_replayed_wal = result + .replay_iterator + .next() + .await + .unwrap() + .expect("expected the replay iterator to contain the fencing WAL"); + assert_eq!( + first_replayed_wal.last_consumed_wal_file_id, + replay_after_wal_id + 1 + ); // verify that fenced db is fenced (new write fails) use crate::error::{CloseReason, ErrorKind}; - let err = db.put(b"k4", b"v4").await.unwrap_err(); + let err = match db.put(b"k4", b"v4").await { + Ok(handle) => handle.await_durable().await.unwrap_err(), + Err(error) => error, + }; assert!( matches!(err.kind(), ErrorKind::Closed(CloseReason::Fenced)), "expected Fenced, got {err}" diff --git a/slatedb/src/filter.rs b/slatedb/src/filter.rs index 8239e22503..453c42087d 100644 --- a/slatedb/src/filter.rs +++ b/slatedb/src/filter.rs @@ -113,7 +113,7 @@ impl BloomFilter { /// checksum, which are accounted for at the SST level. pub(crate) fn estimate_encoded_size(num_keys: u32, filter_bits_per_key: u32) -> usize { let filter_bytes = BloomFilterBuilder::filter_size_bytes(num_keys, filter_bits_per_key); - let num_probes_size = std::mem::size_of::(); + let num_probes_size = size_of::(); filter_bytes + num_probes_size } diff --git a/slatedb/src/filter_iterator.rs b/slatedb/src/filter_iterator.rs index 2ba53508b8..c6ce9f20fc 100644 --- a/slatedb/src/filter_iterator.rs +++ b/slatedb/src/filter_iterator.rs @@ -41,6 +41,8 @@ impl RowEntryIterator for FilterIterator { if (self.predicate)(&entry) { return Ok(Some(entry)); } + // Keep filtered scans cooperative. + tokio::task::coop::consume_budget().await; } Ok(None) } diff --git a/slatedb/src/flatbuffer_types.rs b/slatedb/src/flatbuffer_types.rs index b7f350ab13..77b5951b25 100644 --- a/slatedb/src/flatbuffer_types.rs +++ b/slatedb/src/flatbuffer_types.rs @@ -42,13 +42,17 @@ use crate::flatbuffer_types::root_generated::{ CompactedSsTableViewArgs, Compaction as FbCompaction, CompactionArgs as FbCompactionArgs, CompactionContext as FbCompactionContext, CompactionSpec as FbCompactionSpec, CompactionStatus as FbCompactionStatus, CompactionsV1, CompactionsV1Args, CompressionFormat, - DrainSegmentSpec, DrainSegmentSpecArgs, ManifestV1Args, Segment as FbSegment, - SegmentArgs as FbSegmentArgs, SortedRun as FbSortedRunV1, SortedRunArgs as FbSortedRunV1Args, + DrainSegmentSpec, DrainSegmentSpecArgs, Segment as FbSegment, SegmentArgs as FbSegmentArgs, SortedRunV2, SortedRunV2Args, SstType as FbSstType, Subcompaction as FbSubcompaction, SubcompactionArgs as FbSubcompactionArgs, TieredCompactionContext as FbTieredCompactionContext, TieredCompactionContextArgs as FbTieredCompactionContextArgs, TieredCompactionSpec, TieredCompactionSpecArgs, Ulid as FbUlid, UlidArgs as FbUlidArgs, Uuid, UuidArgs, }; +// V1-only manifest encoder types; only test fixtures build V1 manifests +#[cfg(test)] +use crate::flatbuffer_types::root_generated::{ + ManifestV1Args, SortedRun as FbSortedRunV1, SortedRunArgs as FbSortedRunV1Args, +}; use crate::format::sst::SST_FORMAT_VERSION; use crate::manifest::{ExternalDb, LsmTreeState, Manifest, ManifestCore, Segment}; use crate::partitioned_keyspace::RangePartitionedKeySpace; @@ -197,16 +201,10 @@ pub(crate) struct FlatBufferManifestCodec {} impl ObjectCodec for FlatBufferManifestCodec { fn encode(&self, manifest: &Manifest) -> Bytes { - // RFC-0024 lazy V2 bump: the V1 schema has no `segments` or - // `segment_extractor_name` fields, so writing segmented state - // through the V1 encoder would silently drop it. Pick V2 the - // moment any segmented state is present; databases that never - // configure an extractor keep writing V1. - if Self::requires_v2(manifest) { - Self::create_from_manifest(manifest) - } else { - Self::create_from_manifest_v1(manifest) - } + // RFC-0004 Phase 2 of the manifest V1->V2 rollout: write V2 + // universally, read V1+V2. V1 read support (and the V1 encoder, + // retained below for tests) stays until no V1 manifests remain + Self::create_from_manifest(manifest) } fn decode(&self, bytes: &Bytes) -> Result> { @@ -304,10 +302,7 @@ impl FlatBufferManifestCodec { manifest_sst.visible_range().map(Self::decode_bytes_range), )); } - compacted.push(db_state::SortedRun { - id: manifest_sr.id(), - sst_views: ssts, - }) + compacted.push(db_state::SortedRun::new(manifest_sr.id(), ssts)) } let checkpoints: Vec = manifest .checkpoints() @@ -494,7 +489,7 @@ impl FlatBufferManifestCodec { } fn decode_sorted_runs_v2( - runs: flatbuffers::Vector<'_, flatbuffers::ForwardsUOffset>>, + runs: Vector<'_, ForwardsUOffset>>, sst_lookup: &std::collections::HashMap, ) -> Result, Box> { runs.iter() @@ -504,11 +499,8 @@ impl FlatBufferManifestCodec { .ssts() .iter() .map(|view| Self::decode_compacted_sst_view(&view, sst_lookup)) - .collect::>()?; - Ok(db_state::SortedRun { - id: sr.id(), - sst_views: ssts, - }) + .collect::, _>>()?; + Ok(db_state::SortedRun::new(sr.id(), ssts)) }, ) .collect() @@ -536,8 +528,8 @@ impl FlatBufferManifestCodec { /// and a sorted-runs vector. fn decode_lsm_tree_v2( last_compacted_l0_sst_view_id: Option, - l0: flatbuffers::Vector<'_, flatbuffers::ForwardsUOffset>>, - compacted: flatbuffers::Vector<'_, flatbuffers::ForwardsUOffset>>, + l0: Vector<'_, ForwardsUOffset>>, + compacted: Vector<'_, ForwardsUOffset>>, sst_lookup: &std::collections::HashMap, ) -> Result> { let last_compacted_l0_sst_view_id = last_compacted_l0_sst_view_id.map(|id| id.ulid()); @@ -556,6 +548,10 @@ impl FlatBufferManifestCodec { }) } + /// Retained for tests that construct V1 manifest fixtures to verify + /// decode-side backward compatibility; production code always writes + /// V2 (see [`FlatBufferManifestCodec::encode`]). + #[cfg(test)] pub(crate) fn create_from_manifest_v1(manifest: &Manifest) -> Bytes { let builder = FlatBufferBuilder::new(); let mut db_fb_builder = DbFlatBufferBuilder::new(builder); @@ -567,12 +563,6 @@ impl FlatBufferManifestCodec { let mut db_fb_builder = DbFlatBufferBuilder::new(builder); db_fb_builder.create_manifest(manifest) } - - /// Whether `manifest` carries state that V1 cannot represent: - /// a configured segment extractor, or any named segment. - fn requires_v2(manifest: &Manifest) -> bool { - manifest.core.segment_extractor_name.is_some() || !manifest.core.segments.is_empty() - } } pub(crate) struct FlatBufferCompactionsCodec {} @@ -870,7 +860,7 @@ impl<'b> DbFlatBufferBuilder<'b> { SsTableId::Wal(_) => { unreachable!("cannot pass WAL SST handle to create compacted sst") } - SsTableId::Compacted(ulid) => ulid, + Compacted(ulid) => ulid, }; let compacted_sst_id = self.add_compacted_sst_id(&ulid); let compacted_sst_info = self.add_sst_info(&handle.info); @@ -897,6 +887,9 @@ impl<'b> DbFlatBufferBuilder<'b> { self.builder.create_vector(compacted_ssts.as_ref()) } + /// V1-only; only test fixtures construct V1 manifests now that + /// `encode()` writes V2 universally (see [`FlatBufferManifestCodec::encode`]). + #[cfg(test)] fn add_compacted_sst_from_view( &mut self, view: &SsTableView, @@ -905,7 +898,7 @@ impl<'b> DbFlatBufferBuilder<'b> { SsTableId::Wal(_) => { unreachable!("cannot pass WAL SST handle to create compacted sst from view") } - SsTableId::Compacted(ulid) => ulid, + Compacted(ulid) => ulid, }; let compacted_sst_id = self.add_compacted_sst_id(&ulid); let compacted_sst_info = self.add_sst_info(&view.sst.info); @@ -929,7 +922,7 @@ impl<'b> DbFlatBufferBuilder<'b> { SsTableId::Wal(_) => { unreachable!("cannot pass WAL SST handle to create compacted sst v2") } - SsTableId::Compacted(ulid) => ulid, + Compacted(ulid) => ulid, }; let compacted_sst_id = self.add_compacted_sst_id(&ulid); let compacted_sst_info = self.add_sst_info(&handle.info); @@ -951,7 +944,7 @@ impl<'b> DbFlatBufferBuilder<'b> { SsTableId::Wal(_) => { unreachable!("cannot pass WAL SST handle to create compacted sst view") } - SsTableId::Compacted(ulid) => ulid, + Compacted(ulid) => ulid, }; let sst_id = self.add_compacted_sst_id(&ulid); let visible_range = view.visible_range.as_ref().map(|r| self.add_bytes_range(r)); @@ -982,7 +975,7 @@ impl<'b> DbFlatBufferBuilder<'b> { &mut self, sorted_run: &db_state::SortedRun, ) -> WIPOffset> { - let ssts = self.add_compacted_sst_views(sorted_run.sst_views.iter()); + let ssts = self.add_compacted_sst_views(sorted_run.sst_views().iter()); SortedRunV2::create( &mut self.builder, &SortedRunV2Args { @@ -1044,12 +1037,13 @@ impl<'b> DbFlatBufferBuilder<'b> { self.builder.create_vector(segment_offsets.as_ref()) } + #[cfg(test)] fn add_sorted_run_v1( &mut self, sorted_run: &db_state::SortedRun, ) -> WIPOffset> { let ssts: Vec> = sorted_run - .sst_views + .sst_views() .iter() .map(|view| self.add_compacted_sst_from_view(view)) .collect(); @@ -1063,6 +1057,7 @@ impl<'b> DbFlatBufferBuilder<'b> { ) } + #[cfg(test)] fn add_sorted_runs_v1( &mut self, sorted_runs: &[db_state::SortedRun], @@ -1312,13 +1307,13 @@ impl<'b> DbFlatBufferBuilder<'b> { std::collections::BTreeMap::new(); for tree in core.trees() { for view in tree.l0.iter() { - if let SsTableId::Compacted(ulid) = view.sst.id { + if let Compacted(ulid) = view.sst.id { unique_ssts.entry(ulid).or_insert(&view.sst); } } for sr in tree.compacted.iter() { - for view in sr.sst_views.iter() { - if let SsTableId::Compacted(ulid) = view.sst.id { + for view in sr.sst_views() { + if let Compacted(ulid) = view.sst.id { unique_ssts.entry(ulid).or_insert(&view.sst); } } @@ -1394,6 +1389,7 @@ impl<'b> DbFlatBufferBuilder<'b> { bytes.into() } + #[cfg(test)] fn create_manifest_v1(&mut self, manifest: &Manifest) -> Bytes { let core = &manifest.core; @@ -1671,21 +1667,21 @@ mod tests { new_sst_handle(b"a", Some(BytesRange::from_ref("c"..="d"))), ]); Arc::make_mut(&mut manifest.core.tree).compacted = vec![ - SortedRun { - id: 0, - sst_views: vec![ + SortedRun::new( + 0, + [ new_sst_handle(b"a", None), new_sst_handle(b"d", Some(BytesRange::from_ref("e".."f"))), ], - }, - SortedRun { - id: 0, - sst_views: vec![ + ), + SortedRun::new( + 0, + [ new_sst_handle(b"a", None), new_sst_handle(b"c", Some(BytesRange::from_ref("c"..))), new_sst_handle(b"d", Some(BytesRange::from_ref("e".."f"))), ], - }, + ), ]; let codec = FlatBufferManifestCodec {}; @@ -1721,11 +1717,11 @@ mod tests { let root = Arc::make_mut(&mut manifest.core.tree); root.l0 = (0..16).map(new_sst_view).collect(); root.compacted = (0..4) - .map(|run| SortedRun { - id: run as u32, - sst_views: (0..8) - .map(|offset| new_sst_view(100 + run * 8 + offset)) - .collect(), + .map(|run| { + SortedRun::new( + run as u32, + (0..8).map(|offset| new_sst_view(100 + run * 8 + offset)), + ) }) .collect(); manifest.core.segment_extractor_name = Some("test".to_string()); @@ -1735,12 +1731,10 @@ mod tests { l0: (0..4) .map(|offset| new_sst_view(1_000 + segment * 16 + offset)) .collect(), - compacted: vec![SortedRun { - id: 100 + segment as u32, - sst_views: (0..4) - .map(|offset| new_sst_view(2_000 + segment * 16 + offset)) - .collect(), - }], + compacted: vec![SortedRun::new( + 100 + segment as u32, + (0..4).map(|offset| new_sst_view(2_000 + segment * 16 + offset)), + )], ..Default::default() }; Segment { @@ -1825,10 +1819,14 @@ mod tests { let v1_bytes = bytes.freeze(); codec.decode(&v1_bytes).expect("Should decode V1 manifest"); - // Test encode/decode round-trip (currently writes V1 for forward compatibility) + // Test encode/decode round-trip (writes V2 universally, per RFC-0004 + // Phase 2 of the manifest V1->V2 rollout) let manifest = Manifest::initial(ManifestCore::new()); let encoded = codec.encode(&manifest); - assert_eq!(u16::from_be_bytes([encoded[0], encoded[1]]), 1); + assert_eq!( + u16::from_be_bytes([encoded[0], encoded[1]]), + MANIFEST_FORMAT_VERSION + ); codec .decode(&encoded) .expect("Should decode manifest round-trip"); @@ -1921,10 +1919,7 @@ mod tests { last_compacted_l0_sst_view_id: None, last_compacted_l0_sst_id: None, l0: VecDeque::new(), - compacted: vec![SortedRun { - id: 0, - sst_views: vec![new_sst_view(), new_sst_view()], - }], + compacted: vec![SortedRun::new(0, [new_sst_view(), new_sst_view()])], }), }, Segment { @@ -1933,10 +1928,7 @@ mod tests { last_compacted_l0_sst_view_id: None, last_compacted_l0_sst_id: None, l0: VecDeque::from(vec![new_sst_view(), new_sst_view()]), - compacted: vec![SortedRun { - id: 1, - sst_views: vec![new_sst_view()], - }], + compacted: vec![SortedRun::new(1, [new_sst_view()])], }), }, ]; @@ -2164,7 +2156,7 @@ mod tests { ]; for status in statuses { - let fb_status = super::FbCompactionStatus::from(status); + let fb_status = FbCompactionStatus::from(status); let round_trip = CompactionStatus::from(fb_status); assert_eq!(round_trip, status); } @@ -2176,7 +2168,7 @@ mod tests { let compactions = Compactions::new(1); let bytes = codec.encode(&compactions); - let invalid_version = super::COMPACTIONS_FORMAT_VERSION + 1; + let invalid_version = COMPACTIONS_FORMAT_VERSION + 1; let mut invalid_bytes = bytes.to_vec(); invalid_bytes[0] = (invalid_version >> 8) as u8; invalid_bytes[1] = invalid_version as u8; @@ -2318,9 +2310,9 @@ mod tests { ..Default::default() }, ))]); - Arc::make_mut(&mut manifest.core.tree).compacted = vec![SortedRun { - id: 1, - sst_views: vec![SsTableView::identity(SsTableHandle::new( + Arc::make_mut(&mut manifest.core.tree).compacted = vec![SortedRun::new( + 1, + [SsTableView::identity(SsTableHandle::new( SsTableId::Compacted(ulid::Ulid::new()), SST_FORMAT_VERSION_LATEST, SsTableInfo { @@ -2328,7 +2320,7 @@ mod tests { ..Default::default() }, ))], - }]; + )]; let codec = FlatBufferManifestCodec {}; // when: @@ -2341,7 +2333,7 @@ mod tests { SST_FORMAT_VERSION_LATEST ); assert_eq!( - decoded.core.tree.compacted[0].sst_views[0] + decoded.core.tree.compacted[0].sst_views()[0] .sst .format_version, SST_FORMAT_VERSION_LATEST @@ -2368,9 +2360,9 @@ mod tests { }, ); let l0_ulid = ulid::Ulid::new(); - let l0_id = super::root_generated::Ulid::create( + let l0_id = root_generated::Ulid::create( &mut fbb, - &super::root_generated::UlidArgs { + &root_generated::UlidArgs { high: (l0_ulid.0 >> 64) as u64, low: ((l0_ulid.0 << 64) >> 64) as u64, }, @@ -2395,9 +2387,9 @@ mod tests { }, ); let sr_ulid = ulid::Ulid::new(); - let sr_id = super::root_generated::Ulid::create( + let sr_id = root_generated::Ulid::create( &mut fbb, - &super::root_generated::UlidArgs { + &root_generated::UlidArgs { high: (sr_ulid.0 >> 64) as u64, low: ((sr_ulid.0 << 64) >> 64) as u64, }, @@ -2420,8 +2412,8 @@ mod tests { }, ); let compacted_vec = fbb.create_vector(&[sorted_run]); - let checkpoints_vec = fbb - .create_vector::>(&[]); + let checkpoints_vec = + fbb.create_vector::>(&[]); let manifest = ManifestV1::create( &mut fbb, &ManifestV1Args { @@ -2447,7 +2439,7 @@ mod tests { super::ORIGINAL_SST_FORMAT_VERSION ); assert_eq!( - decoded.core.tree.compacted[0].sst_views[0] + decoded.core.tree.compacted[0].sst_views()[0] .sst .format_version, super::ORIGINAL_SST_FORMAT_VERSION @@ -2518,9 +2510,9 @@ mod tests { }, ); let sst_ulid = ulid::Ulid::new(); - let sst_id = super::root_generated::Ulid::create( + let sst_id = root_generated::Ulid::create( &mut fbb, - &super::root_generated::UlidArgs { + &root_generated::UlidArgs { high: (sst_ulid.0 >> 64) as u64, low: ((sst_ulid.0 << 64) >> 64) as u64, }, @@ -2537,9 +2529,9 @@ mod tests { let output_ssts_vec = fbb.create_vector(&[output_sst]); // Build a compaction with a tiered spec let source_ulid = ulid::Ulid::new(); - let source_id = super::root_generated::Ulid::create( + let source_id = root_generated::Ulid::create( &mut fbb, - &super::root_generated::UlidArgs { + &root_generated::UlidArgs { high: (source_ulid.0 >> 64) as u64, low: ((source_ulid.0 << 64) >> 64) as u64, }, @@ -2557,9 +2549,9 @@ mod tests { }, ); let compaction_ulid = ulid::Ulid::new(); - let compaction_id = super::root_generated::Ulid::create( + let compaction_id = root_generated::Ulid::create( &mut fbb, - &super::root_generated::UlidArgs { + &root_generated::UlidArgs { high: (compaction_ulid.0 >> 64) as u64, low: ((compaction_ulid.0 << 64) >> 64) as u64, }, @@ -2573,7 +2565,7 @@ mod tests { status: fb_status, output_ssts: Some(output_ssts_vec), worker: None, - ctx_type: super::root_generated::CompactionContext::NONE, + ctx_type: root_generated::CompactionContext::NONE, ctx: None, }, ); @@ -2614,9 +2606,9 @@ mod tests { }; let mut fbb = flatbuffers::FlatBufferBuilder::new(); let source_ulid = ulid::Ulid::new(); - let source_id = super::root_generated::Ulid::create( + let source_id = root_generated::Ulid::create( &mut fbb, - &super::root_generated::UlidArgs { + &root_generated::UlidArgs { high: (source_ulid.0 >> 64) as u64, low: ((source_ulid.0 << 64) >> 64) as u64, }, @@ -2633,9 +2625,9 @@ mod tests { }, ); let compaction_ulid = ulid::Ulid::new(); - let compaction_id = super::root_generated::Ulid::create( + let compaction_id = root_generated::Ulid::create( &mut fbb, - &super::root_generated::UlidArgs { + &root_generated::UlidArgs { high: (compaction_ulid.0 >> 64) as u64, low: ((compaction_ulid.0 << 64) >> 64) as u64, }, @@ -2649,7 +2641,7 @@ mod tests { status: FbCompactionStatus::Running, output_ssts: None, worker: None, - ctx_type: super::root_generated::CompactionContext::NONE, + ctx_type: root_generated::CompactionContext::NONE, ctx: None, }, ); @@ -2701,9 +2693,9 @@ mod tests { }, ); let compaction_ulid = ulid::Ulid::new(); - let compaction_id = super::root_generated::Ulid::create( + let compaction_id = root_generated::Ulid::create( &mut fbb, - &super::root_generated::UlidArgs { + &root_generated::UlidArgs { high: (compaction_ulid.0 >> 64) as u64, low: ((compaction_ulid.0 << 64) >> 64) as u64, }, @@ -2717,7 +2709,7 @@ mod tests { status: FbCompactionStatus::Running, output_ssts: None, worker: None, - ctx_type: super::root_generated::CompactionContext::NONE, + ctx_type: root_generated::CompactionContext::NONE, ctx: None, }, ); @@ -2769,20 +2761,20 @@ mod tests { new_view(b"b", Some(BytesRange::from_ref("c"..="d"))), ]); Arc::make_mut(&mut manifest.core.tree).compacted = vec![ - SortedRun { - id: 1, - sst_views: vec![ + SortedRun::new( + 1, + [ new_view(b"e", None), new_view(b"f", Some(BytesRange::from_ref("g".."h"))), ], - }, - SortedRun { - id: 2, - sst_views: vec![ + ), + SortedRun::new( + 2, + [ new_view(b"i", None), new_view(b"j", Some(BytesRange::from_ref("k"..))), ], - }, + ), ]; Arc::make_mut(&mut manifest.core.tree).last_compacted_l0_sst_view_id = Some(manifest.core.tree.l0[0].id); @@ -2843,9 +2835,9 @@ mod tests { ..Default::default() }, ))]); - Arc::make_mut(&mut manifest.core.tree).compacted = vec![SortedRun { - id: 1, - sst_views: vec![SsTableView::identity(SsTableHandle::new( + Arc::make_mut(&mut manifest.core.tree).compacted = vec![SortedRun::new( + 1, + [SsTableView::identity(SsTableHandle::new( SsTableId::Compacted(ulid::Ulid::new()), SST_FORMAT_VERSION_LATEST, SsTableInfo { @@ -2853,7 +2845,7 @@ mod tests { ..Default::default() }, ))], - }]; + )]; manifest.writer_epoch = 5; manifest.compactor_epoch = 3; diff --git a/slatedb/src/flush.rs b/slatedb/src/flush.rs index 2563f2a608..bdd5e1e135 100644 --- a/slatedb/src/flush.rs +++ b/slatedb/src/flush.rs @@ -37,6 +37,8 @@ impl DbInner { while let Some(entry) = iter.next().await? { sst_builder.add(entry).await?; any = true; + // Keep cached flush work cooperative. + tokio::task::coop::consume_budget().await; } if !any { return Ok(None); @@ -131,6 +133,8 @@ impl DbInner { } current_builder.add(entry).await?; current_has_entry = true; + // Keep cached flush work cooperative. + tokio::task::coop::consume_budget().await; } if current_has_entry { out.push(EncodedSegmentSst { diff --git a/slatedb/src/format/block.rs b/slatedb/src/format/block.rs index 9647f39dc2..a3b6944352 100644 --- a/slatedb/src/format/block.rs +++ b/slatedb/src/format/block.rs @@ -5,7 +5,7 @@ use crate::types::RowEntry; use crate::utils::clamp_allocated_size_bytes; use bytes::{BufMut, Bytes, BytesMut}; -pub(crate) const SIZEOF_U16: usize = std::mem::size_of::(); +pub(crate) const SIZEOF_U16: usize = size_of::(); #[derive(Eq, PartialEq)] pub(crate) struct Block { @@ -72,11 +72,8 @@ impl Block { } let offsets_raw = &data[data_end..data.len() - SIZEOF_U16]; let mut offsets = Vec::with_capacity(entry_offsets_len); - for raw in offsets_raw.chunks_exact(SIZEOF_U16) { - let offset = u16::from_be_bytes( - raw.try_into() - .map_err(|_| corrupt_block("invalid block row offset"))?, - ); + for raw in offsets_raw.as_chunks::().0 { + let offset = u16::from_be_bytes(*raw); if usize::from(offset) >= data_end { return Err(corrupt_block("block row offset is outside row data")); } @@ -141,7 +138,9 @@ fn compute_prefix(lhs: &[u8], rhs: &[u8]) -> usize { } fn compute_prefix_chunks(lhs: &[u8], rhs: &[u8]) -> usize { - let off = std::iter::zip(lhs.chunks_exact(N), rhs.chunks_exact(N)) + let (lhs_chunks, _) = lhs.as_chunks::(); + let (rhs_chunks, _) = rhs.as_chunks::(); + let off = std::iter::zip(lhs_chunks, rhs_chunks) .take_while(|(a, b)| a == b) .count() * N; diff --git a/slatedb/src/format/block_v2.rs b/slatedb/src/format/block_v2.rs index 06de4d2382..9e88977bd5 100644 --- a/slatedb/src/format/block_v2.rs +++ b/slatedb/src/format/block_v2.rs @@ -64,7 +64,9 @@ fn compute_prefix(lhs: &[u8], rhs: &[u8]) -> usize { /// the overhead of chunk iteration and the benefits of bulk comparison. fn compute_prefix_chunks(lhs: &[u8], rhs: &[u8]) -> usize { // Compare N-byte chunks until we find one that differs - let off = std::iter::zip(lhs.chunks_exact(N), rhs.chunks_exact(N)) + let (lhs_chunks, _) = lhs.as_chunks::(); + let (rhs_chunks, _) = rhs.as_chunks::(); + let off = std::iter::zip(lhs_chunks, rhs_chunks) .take_while(|(a, b)| a == b) .count() * N; diff --git a/slatedb/src/format/row.rs b/slatedb/src/format/row.rs index 8bcd0a8632..a69d7ab896 100644 --- a/slatedb/src/format/row.rs +++ b/slatedb/src/format/row.rs @@ -147,10 +147,10 @@ impl SstRowCodecV0 { /// estimated_entries_size include the size of seqnum,create_ts(if exist),expire_ts(exist),key,value pub(crate) fn estimate_encoded_size(entry_num: usize, estimated_entries_size: usize) -> usize { - let key_prefix_len_size = std::mem::size_of::(); - let key_suffix_len_size = std::mem::size_of::(); - let value_len_size = std::mem::size_of::(); - let flag_size = std::mem::size_of::(); + let key_prefix_len_size = size_of::(); + let key_suffix_len_size = size_of::(); + let value_len_size = size_of::(); + let flag_size = size_of::(); let mut ans = estimated_entries_size; ans += (key_prefix_len_size + key_suffix_len_size + value_len_size + flag_size) * entry_num; ans diff --git a/slatedb/src/format/sst.rs b/slatedb/src/format/sst.rs index 0589f35808..c69e2b9c6b 100644 --- a/slatedb/src/format/sst.rs +++ b/slatedb/src/format/sst.rs @@ -141,10 +141,7 @@ impl BlockBuilder { } } - pub(crate) fn add( - &mut self, - entry: crate::types::RowEntry, - ) -> Result { + pub(crate) fn add(&mut self, entry: crate::types::RowEntry) -> Result { match self { Self::V1(builder) => builder.add(entry), Self::V2(builder) => builder.add(entry), @@ -197,10 +194,7 @@ impl BlockBuilderWithStats { self.builder.would_fit(entry) } - pub(crate) fn add( - &mut self, - entry: crate::types::RowEntry, - ) -> Result { + pub(crate) fn add(&mut self, entry: crate::types::RowEntry) -> Result { match &entry.value { crate::types::ValueDeletable::Value(_) => self.stats.num_puts += 1, crate::types::ValueDeletable::Merge(_) => self.stats.num_merges += 1, @@ -414,7 +408,7 @@ pub(crate) struct EncodedSsTableFooterBuilder<'a, 'b> { /// codec for the SST info sst_info_codec: &'a dyn SsTableInfoCodec, /// builder for the index block - index_builder: flatbuffers::FlatBufferBuilder<'b, flatbuffers::DefaultAllocator>, + index_builder: flatbuffers::FlatBufferBuilder<'b, DefaultAllocator>, /// metadata block block_meta: Vec>>, /// filter blocks @@ -621,7 +615,12 @@ pub(crate) async fn compress_and_transform( ) -> Result { let compressed = match compression_codec { None => data, - Some(c) => compress(data, c)?, + Some(c) => { + let compressed = compress(data, c)?; + // Account for CPU-only compression work. + tokio::task::coop::consume_budget().await; + compressed + } }; let transformed = transform(compressed, block_transformer).await?; let checksum = crc32fast::hash(&transformed); @@ -685,10 +684,15 @@ pub(crate) async fn transform( block_transformer: Option<&Arc>, ) -> Result { let transformed = match block_transformer { - Some(t) => t - .encode(data) - .await - .map_err(|_| SlateDBError::BlockTransformError)?, + Some(t) => { + let transformed = t + .encode(data) + .await + .map_err(|_| SlateDBError::BlockTransformError)?; + // Account for CPU-only transformation work. + tokio::task::coop::consume_budget().await; + transformed + } None => data, }; Ok(transformed) @@ -698,6 +702,7 @@ pub(crate) type LengthOffsetAndVersion = (u64, u64, u16); pub(crate) type TableInfoAndVersion = (SsTableInfo, u16); +#[allow(dead_code)] pub(crate) enum StagedSstInfoError { Initial(SlateDBError), Later(SlateDBError), @@ -728,6 +733,7 @@ impl Default for SsTableFormat { } } +#[allow(dead_code)] impl SsTableFormat { async fn read_length_and_metadata_offset_and_version( &self, @@ -1247,13 +1253,17 @@ impl SsTableFormat { } } - fn block_range( + pub(crate) fn block_range( &self, blocks: Range, info: &SsTableInfo, index: &SsTableIndex, ) -> Range { - let mut end_offset = info.filter_offset; + let mut end_offset = if info.filter_len > 0 { + info.filter_offset + } else { + info.index_offset + }; if blocks.end < index.block_meta().len() { let next_block_meta = index.block_meta().get(blocks.end); end_offset = next_block_meta.offset(); diff --git a/slatedb/src/garbage_collector.rs b/slatedb/src/garbage_collector.rs index 426e235dcd..4ab9d4f7a5 100644 --- a/slatedb/src/garbage_collector.rs +++ b/slatedb/src/garbage_collector.rs @@ -24,6 +24,7 @@ use crate::manifest::store::{ManifestStore, StoredManifest}; use crate::manifest::Manifest; use crate::tablestore::TableStore; use crate::utils::WatchableOnceCell; +use crate::wal::slatedb::gc::{SlateDbWalGc, WalGcMode}; use async_trait::async_trait; use chrono::{DateTime, Utc}; use compacted_gc::CompactedGcTask; @@ -40,7 +41,7 @@ use std::sync::Arc; use std::time::Duration; use tokio::runtime::Handle; use tracing::instrument; -use wal_gc::{WalGcMode, WalGcTask}; +use wal_gc::WalGcTask; mod compacted_gc; mod compactions_gc; @@ -50,10 +51,12 @@ mod manifest_gc; pub mod stats; mod wal_gc; +use crate::wal::WalGc; +pub(crate) use filter::retain_allowed_by_gc_filter; pub use filter::GcFilter; pub(crate) const DEFAULT_MIN_AGE: Duration = Duration::from_secs(300); -pub(crate) const DEFAULT_INTERVAL: Duration = Duration::from_secs(60); +pub(crate) const DEFAULT_INTERVAL: Duration = Duration::from_secs(600); pub(crate) const GC_TASK_NAME: &str = "garbage_collector"; /// Maximum number of concurrent object-store deletes issued by a GC task's /// deletion pass. Deletes are independent single-object operations, so a small @@ -233,6 +236,7 @@ impl GarbageCollector { recorder: &MetricsRecorderHelper, system_clock: Arc, gc_filter: Option>, + wal_gc: Option>, ) -> Self { let stats = Arc::new(GcStats::new(recorder)); // The standalone GC lifecycle does not surface a closed result yet, so the @@ -243,23 +247,37 @@ impl GarbageCollector { system_clock.clone(), )); let wal_gc_task = options.wal_options.map(|wal_options| { + let wal_gc = wal_gc.unwrap_or_else(|| { + Arc::new(SlateDbWalGc::new( + table_store.clone(), + stats.clone(), + WalGcMode::Regular, + gc_filter.clone(), + system_clock.clone(), + )) + }); WalGcTask::new( manifest_store.clone(), - table_store.clone(), - stats.clone(), - wal_options, - WalGcMode::Regular, - gc_filter.clone(), + wal_gc, + WalGcMode::Regular.resource(), + wal_options.min_age, + wal_options.dry_run, ) }); let wal_fence_gc_task = options.wal_fence_options.map(|wal_fence_options| { - WalGcTask::new( - manifest_store.clone(), + let wal_gc = Arc::new(SlateDbWalGc::new( table_store.clone(), stats.clone(), - wal_fence_options, WalGcMode::Fence, gc_filter.clone(), + system_clock.clone(), + )); + WalGcTask::new( + manifest_store.clone(), + wal_gc, + WalGcMode::Fence.resource(), + wal_fence_options.min_age, + wal_fence_options.dry_run, ) }); let compacted_gc_task = options.compacted_options.map(|compacted_options| { @@ -692,7 +710,7 @@ mod tests { fn new_checkpoint(manifest_id: u64, expire_time: Option>) -> Checkpoint { Checkpoint { - id: uuid::Uuid::new_v4(), + id: Uuid::new_v4(), manifest_id, expire_time, create_time: DefaultSystemClock::default().now(), @@ -1182,6 +1200,7 @@ mod tests { &MetricsRecorderHelper::noop(), Arc::new(DefaultSystemClock::default()), None, + None, ); gc.run_gc_once().await; @@ -1252,6 +1271,7 @@ mod tests { &helper, Arc::new(DefaultSystemClock::default()), None, + None, ); gc.run_gc_once().await; @@ -1263,7 +1283,7 @@ mod tests { assert_eq!( lookup_metric_with_labels( &recorder, - crate::garbage_collector::stats::DELETED_COUNT, + stats::DELETED_COUNT, &[("resource", "wal_fence")] ), Some(2) @@ -1317,6 +1337,7 @@ mod tests { &MetricsRecorderHelper::noop(), Arc::new(DefaultSystemClock::default()), None, + None, ); gc.run_gc_once().await; @@ -1397,6 +1418,7 @@ mod tests { &MetricsRecorderHelper::noop(), Arc::new(DefaultSystemClock::default()), None, + None, ); gc.run_gc_once().await; @@ -1449,14 +1471,16 @@ mod tests { .l0 .push_back(SsTableView::identity(active_expired_l0_sst_handle.clone())); // Dont' push inactive_expired_l0_sst_handle - Arc::make_mut(&mut state.tree).compacted.push(SortedRun { - id: 1, - // Don't add inactive_expired_sst_handle - sst_views: vec![ - SsTableView::identity(active_sst_handle.clone()), - SsTableView::identity(active_expired_sst_handle.clone()), - ], - }); + Arc::make_mut(&mut state.tree) + .compacted + .push(SortedRun::new( + 1, + // Don't add inactive_expired_sst_handle + [ + SsTableView::identity(active_sst_handle.clone()), + SsTableView::identity(active_expired_sst_handle.clone()), + ], + )); StoredManifest::create_new_db( manifest_store.clone(), state.clone(), @@ -1488,7 +1512,7 @@ mod tests { assert_eq!(current_manifest.manifest.core.tree.compacted.len(), 1); assert_eq!( current_manifest.manifest.core.tree.compacted[0] - .sst_views + .sst_views() .len(), 2 ); @@ -1525,7 +1549,7 @@ mod tests { assert_eq!(current_manifest.manifest.core.tree.compacted.len(), 1); assert_eq!( current_manifest.manifest.core.tree.compacted[0] - .sst_views + .sst_views() .len(), 2 ); @@ -1569,14 +1593,18 @@ mod tests { .push_back(SsTableView::identity( active_checkpoint_l0_sst_handle.clone(), )); - Arc::make_mut(&mut state.tree).compacted.push(SortedRun { - id: 1, - sst_views: vec![SsTableView::identity(active_sst_handle.clone())], - }); - Arc::make_mut(&mut state.tree).compacted.push(SortedRun { - id: 2, - sst_views: vec![SsTableView::identity(active_checkpoint_sst_handle.clone())], - }); + Arc::make_mut(&mut state.tree) + .compacted + .push(SortedRun::new( + 1, + [SsTableView::identity(active_sst_handle.clone())], + )); + Arc::make_mut(&mut state.tree) + .compacted + .push(SortedRun::new( + 2, + [SsTableView::identity(active_checkpoint_sst_handle.clone())], + )); let mut stored_manifest = StoredManifest::create_new_db( manifest_store.clone(), state.clone(), @@ -1756,7 +1784,7 @@ mod tests { } for sr in &manifest.core.tree.compacted { - for view in &sr.sst_views { + for view in sr.sst_views() { assert!(compacted_ssts.contains(&view.sst.id)); } } @@ -1820,23 +1848,23 @@ mod tests { let gc_opts = GarbageCollectorOptions { manifest_options: Some(GarbageCollectorDirectoryOptions { - min_age: std::time::Duration::from_secs(3600), + min_age: Duration::from_secs(3600), interval: None, dry_run: false, }), - wal_options: Some(crate::config::GarbageCollectorDirectoryOptions { - min_age: std::time::Duration::from_secs(3600), + wal_options: Some(GarbageCollectorDirectoryOptions { + min_age: Duration::from_secs(3600), interval: None, dry_run: false, }), wal_fence_options: None, - compacted_options: Some(crate::config::GarbageCollectorDirectoryOptions { - min_age: std::time::Duration::from_secs(3600), + compacted_options: Some(GarbageCollectorDirectoryOptions { + min_age: Duration::from_secs(3600), interval: None, dry_run: false, }), - compactions_options: Some(crate::config::GarbageCollectorDirectoryOptions { - min_age: std::time::Duration::from_secs(3600), + compactions_options: Some(GarbageCollectorDirectoryOptions { + min_age: Duration::from_secs(3600), interval: None, dry_run: false, }), @@ -1855,6 +1883,7 @@ mod tests { recorder, Arc::new(DefaultSystemClock::default()), None, + None, ); gc.run_gc_once().await; @@ -1897,23 +1926,23 @@ mod tests { let recorder = MetricsRecorderHelper::noop(); let gc_opts = GarbageCollectorOptions { manifest_options: Some(GarbageCollectorDirectoryOptions { - min_age: std::time::Duration::from_secs(3600), + min_age: Duration::from_secs(3600), interval: None, dry_run: false, }), wal_options: Some(GarbageCollectorDirectoryOptions { - min_age: std::time::Duration::from_secs(3600), + min_age: Duration::from_secs(3600), interval: None, dry_run: false, }), wal_fence_options: None, compacted_options: Some(GarbageCollectorDirectoryOptions { - min_age: std::time::Duration::from_secs(3600), + min_age: Duration::from_secs(3600), interval: None, dry_run: false, }), compactions_options: Some(GarbageCollectorDirectoryOptions { - min_age: std::time::Duration::from_secs(3600), + min_age: Duration::from_secs(3600), interval: None, dry_run: false, }), @@ -1932,6 +1961,7 @@ mod tests { &recorder, Arc::new(DefaultSystemClock::default()), None, + None, ); // Send a WAL GC message. Correct behavior: only WAL GC runs. @@ -1974,18 +2004,18 @@ mod tests { let gc_opts = GarbageCollectorOptions { manifest_options: None, wal_options: Some(GarbageCollectorDirectoryOptions { - min_age: std::time::Duration::from_secs(3600), + min_age: Duration::from_secs(3600), interval: None, dry_run: false, }), wal_fence_options: None, compacted_options: Some(GarbageCollectorDirectoryOptions { - min_age: std::time::Duration::from_secs(3600), + min_age: Duration::from_secs(3600), interval: None, dry_run: false, }), compactions_options: Some(GarbageCollectorDirectoryOptions { - min_age: std::time::Duration::from_secs(3600), + min_age: Duration::from_secs(3600), interval: None, dry_run: false, }), @@ -2004,6 +2034,7 @@ mod tests { &recorder, Arc::new(DefaultSystemClock::default()), None, + None, ); gc.run_gc_once().await; @@ -2055,6 +2086,7 @@ mod tests { &recorder, Arc::new(DefaultSystemClock::default()), None, + None, ); let intervals: Vec<_> = gc.tickers().into_iter().map(|def| def.interval).collect(); @@ -2079,18 +2111,18 @@ mod tests { interval: Some(Duration::from_secs(1)), dry_run: false, }), - wal_options: Some(crate::config::GarbageCollectorDirectoryOptions { + wal_options: Some(GarbageCollectorDirectoryOptions { min_age: Duration::from_secs(3600), interval: Some(Duration::from_secs(1)), dry_run: false, }), wal_fence_options: None, - compacted_options: Some(crate::config::GarbageCollectorDirectoryOptions { + compacted_options: Some(GarbageCollectorDirectoryOptions { min_age: Duration::from_secs(3600), interval: Some(Duration::from_secs(1)), dry_run: false, }), - compactions_options: Some(crate::config::GarbageCollectorDirectoryOptions { + compactions_options: Some(GarbageCollectorDirectoryOptions { min_age: Duration::from_secs(3600), interval: Some(Duration::from_secs(1)), dry_run: false, @@ -2110,6 +2142,7 @@ mod tests { &recorder, Arc::new(DefaultSystemClock::default()), None, + None, ); gc.start().expect("failed to start garbage collector"); gc.stop().await.expect("failed to stop garbage collector"); @@ -2151,11 +2184,7 @@ mod tests { // then: assert_eq!( - lookup_metric_with_labels( - &recorder, - crate::garbage_collector::stats::DELETED_COUNT, - &[("resource", "manifest")] - ), + lookup_metric_with_labels(&recorder, stats::DELETED_COUNT, &[("resource", "manifest")]), Some(1) ); } @@ -2196,11 +2225,7 @@ mod tests { // then: assert_eq!( - lookup_metric_with_labels( - &recorder, - crate::garbage_collector::stats::DELETED_COUNT, - &[("resource", "wal")] - ), + lookup_metric_with_labels(&recorder, stats::DELETED_COUNT, &[("resource", "wal")]), Some(1) ); } @@ -2221,10 +2246,9 @@ mod tests { Arc::make_mut(&mut state.tree) .l0 .push_back(SsTableView::identity(active_l0_handle)); - Arc::make_mut(&mut state.tree).compacted.push(SortedRun { - id: 1, - sst_views: vec![SsTableView::identity(active_handle)], - }); + Arc::make_mut(&mut state.tree) + .compacted + .push(SortedRun::new(1, [SsTableView::identity(active_handle)])); // inactive_expired_handle is NOT in manifest -> eligible for GC StoredManifest::create_new_db( manifest_store.clone(), @@ -2258,7 +2282,7 @@ mod tests { assert_eq!( lookup_metric_with_labels( &recorder, - crate::garbage_collector::stats::DELETED_COUNT, + stats::DELETED_COUNT, &[("resource", "compacted")] ), Some(1) @@ -2322,7 +2346,7 @@ mod tests { assert_eq!( lookup_metric_with_labels( &recorder, - crate::garbage_collector::stats::DELETED_COUNT, + stats::DELETED_COUNT, &[("resource", "compactions")] ), Some(2) @@ -2444,6 +2468,7 @@ mod tests { Some(Arc::new(LocationGcFilter { allowed_locations: HashSet::new(), })), + None, ); // Run every directory GC task with candidates present for each task type. @@ -2545,6 +2570,7 @@ mod tests { Some(Arc::new(LocationGcFilter { allowed_locations: HashSet::from([path_resolver.sst_path(&allowed_wal_id)]), })), + None, ); // Run WAL GC with a filter that permits only one of the eligible WALs. @@ -2560,11 +2586,7 @@ mod tests { .collect::>(); assert_eq!(wal_ids, vec![rejected_before_wal_id, rejected_after_wal_id]); assert_eq!( - lookup_metric_with_labels( - &recorder, - crate::garbage_collector::stats::DELETED_COUNT, - &[("resource", "wal")] - ), + lookup_metric_with_labels(&recorder, stats::DELETED_COUNT, &[("resource", "wal")]), Some(1) ); } @@ -2677,6 +2699,7 @@ mod tests { &recorder, Arc::new(DefaultSystemClock::default()), None, + None, ); gc.run_gc_once().await; diff --git a/slatedb/src/garbage_collector/compacted_gc.rs b/slatedb/src/garbage_collector/compacted_gc.rs index de1570f663..9009e91e1a 100644 --- a/slatedb/src/garbage_collector/compacted_gc.rs +++ b/slatedb/src/garbage_collector/compacted_gc.rs @@ -149,7 +149,7 @@ fn collect_active_ssts<'a>(manifests: impl Iterator) -> Has active.insert(view.sst.id); } for sr in tree.compacted.iter() { - for view in sr.sst_views.iter() { + for view in sr.sst_views() { active.insert(view.sst.id); } } @@ -741,10 +741,7 @@ mod tests { last_compacted_l0_sst_view_id: None, last_compacted_l0_sst_id: None, l0: VecDeque::from(vec![segment_l0.clone()]), - compacted: vec![SortedRun { - id: 0, - sst_views: vec![segment_sr.clone()], - }], + compacted: vec![SortedRun::new(0, [segment_sr.clone()])], }), }], ); diff --git a/slatedb/src/garbage_collector/compactions_gc.rs b/slatedb/src/garbage_collector/compactions_gc.rs index f4634dd2c0..326e448bd5 100644 --- a/slatedb/src/garbage_collector/compactions_gc.rs +++ b/slatedb/src/garbage_collector/compactions_gc.rs @@ -25,6 +25,7 @@ use crate::{ use chrono::{DateTime, Utc}; use futures::StreamExt; use log::error; +use std::sync::atomic::{AtomicU64, Ordering}; use std::sync::Arc; use super::filter::retain_allowed_by_gc_filter; @@ -72,7 +73,9 @@ impl CompactionsGcTask { /// Deletes the given compactions files from the compactions store. /// /// In case of dryrun, the actual deletion doesn't happen. - async fn maybe_delete_compactions(&self, compactions_ids: Vec) { + /// + /// Returns the number of compactions files actually deleted. + async fn maybe_delete_compactions(&self, compactions_ids: Vec) -> u64 { if self.compactions_options.dry_run { if !compactions_ids.is_empty() { log::info!( @@ -86,22 +89,28 @@ impl CompactionsGcTask { id ); } - return; + return 0; } + let deleted_count = AtomicU64::new(0); futures::stream::iter(compactions_ids) - .for_each_concurrent(GC_DELETE_CONCURRENCY, |id| async move { - if let Err(e) = self - .compactions_store - .delete_compactions_unchecked(id) - .await - { - error!("error deleting compactions [id={:?}, error={}]", id, e); - } else { - self.stats.gc_compactions_count.increment(1); + .for_each_concurrent(GC_DELETE_CONCURRENCY, |id| { + let deleted_count = &deleted_count; + async move { + if let Err(e) = self + .compactions_store + .delete_compactions_unchecked(id) + .await + { + error!("error deleting compactions [id={:?}, error={}]", id, e); + } else { + self.stats.gc_compactions_count.increment(1); + deleted_count.fetch_add(1, Ordering::Relaxed); + } } }) .await; + deleted_count.load(Ordering::Relaxed) } } @@ -112,6 +121,7 @@ impl GcTask for CompactionsGcTask { async fn collect(&self, utc_now: DateTime) -> Result<(), SlateDBError> { let min_age = self.compactions_min_age(); let mut compactions_metadata_list = self.compactions_store.list_compactions(..).await?; + let pre_gc_count = compactions_metadata_list.len() as u64; // Remove the last element so we never delete the latest compactions file compactions_metadata_list.pop(); @@ -142,9 +152,18 @@ impl GcTask for CompactionsGcTask { .map(|compactions_metadata| compactions_metadata.id) .collect::>(); - self.maybe_delete_compactions(compactions_ids_to_delete) + self.stats.gc_compactions_versions.set(pre_gc_count as i64); + + let deleted_count = self + .maybe_delete_compactions(compactions_ids_to_delete) .await; + if deleted_count > 0 { + self.stats + .gc_compactions_versions + .set((pre_gc_count - deleted_count) as i64); + } + Ok(()) } @@ -160,7 +179,9 @@ mod tests { use async_trait::async_trait; use chrono::TimeDelta; use object_store::{memory::InMemory, path::Path, ObjectStoreExt}; - use slatedb_common::metrics::MetricsRecorderHelper; + use slatedb_common::metrics::{ + lookup_metric_with_labels, DefaultMetricsRecorder, MetricsRecorderHelper, + }; use slatedb_common::ObjectMetadata; use std::collections::HashSet; use std::time::Duration; @@ -337,4 +358,146 @@ mod tests { .unwrap() .is_some()); } + + async fn make_compactions_store() -> (Arc, StoredCompactions) { + let object_store = Arc::new(InMemory::new()); + let compactions_store = Arc::new(CompactionsStore::new(&Path::from("/root"), object_store)); + let stored = StoredCompactions::create(compactions_store.clone(), 0) + .await + .unwrap(); + (compactions_store, stored) + } + + #[tokio::test] + async fn test_version_count_after_gc_deletes_old_compactions() { + let (compactions_store, mut stored) = make_compactions_store().await; + // Write two more compactions files: ids 1 (create), 2, 3 + stored + .update(stored.prepare_dirty().unwrap()) + .await + .unwrap(); + stored + .update(stored.prepare_dirty().unwrap()) + .await + .unwrap(); + + let metrics = Arc::new(DefaultMetricsRecorder::new()); + let recorder = MetricsRecorderHelper::new(metrics.clone(), Default::default()); + let task = CompactionsGcTask::new( + compactions_store.clone(), + Arc::new(GcStats::new(&recorder)), + GarbageCollectorDirectoryOptions { + min_age: Duration::ZERO, + interval: None, + dry_run: false, + }, + None, + true, + ); + + task.collect(Utc::now() + TimeDelta::hours(1)) + .await + .unwrap(); + + // GC deletes compactions 1 and 2 (older than min_age=0); compactions 3 survives as latest. + assert_eq!( + lookup_metric_with_labels( + &metrics, + crate::garbage_collector::stats::VERSION_COUNT, + &[("resource", "compactions")] + ), + Some(1), + "expected 1 surviving compactions file after GC" + ); + } + + #[tokio::test] + async fn test_version_count_when_nothing_to_delete() { + let (compactions_store, mut stored) = make_compactions_store().await; + stored + .update(stored.prepare_dirty().unwrap()) + .await + .unwrap(); + stored + .update(stored.prepare_dirty().unwrap()) + .await + .unwrap(); + + let metrics = Arc::new(DefaultMetricsRecorder::new()); + let recorder = MetricsRecorderHelper::new(metrics.clone(), Default::default()); + let task = CompactionsGcTask::new( + compactions_store.clone(), + Arc::new(GcStats::new(&recorder)), + GarbageCollectorDirectoryOptions { + min_age: Duration::from_secs(3600), // too new to delete + interval: None, + dry_run: false, + }, + None, + true, + ); + + task.collect(Utc::now()).await.unwrap(); + + // Nothing deleted — all 3 compactions files survive. + assert_eq!( + lookup_metric_with_labels( + &metrics, + crate::garbage_collector::stats::VERSION_COUNT, + &[("resource", "compactions")] + ), + Some(3), + "expected all 3 compactions files when nothing qualifies for deletion" + ); + } + + #[tokio::test] + async fn test_version_count_unchanged_on_dry_run() { + let (compactions_store, mut stored) = make_compactions_store().await; + // Write two more compactions files: ids 1 (create), 2, 3 + stored + .update(stored.prepare_dirty().unwrap()) + .await + .unwrap(); + stored + .update(stored.prepare_dirty().unwrap()) + .await + .unwrap(); + + let metrics = Arc::new(DefaultMetricsRecorder::new()); + let recorder = MetricsRecorderHelper::new(metrics.clone(), Default::default()); + let task = CompactionsGcTask::new( + compactions_store.clone(), + Arc::new(GcStats::new(&recorder)), + GarbageCollectorDirectoryOptions { + min_age: Duration::ZERO, + interval: None, + dry_run: true, + }, + None, + true, + ); + + task.collect(Utc::now() + TimeDelta::hours(1)) + .await + .unwrap(); + + // Dry run deletes nothing, so all 3 compactions files still exist and the gauge + // reports the true current count rather than the hypothetical post-deletion count. + assert_eq!( + lookup_metric_with_labels( + &metrics, + crate::garbage_collector::stats::VERSION_COUNT, + &[("resource", "compactions")] + ), + Some(3), + "expected dry run to leave the gauge at the true current count" + ); + let compactions = compactions_store.list_compactions(..).await.unwrap(); + assert_eq!( + compactions.len(), + 3, + "dry run should not delete any compactions files" + ); + } } diff --git a/slatedb/src/garbage_collector/manifest_gc.rs b/slatedb/src/garbage_collector/manifest_gc.rs index 709ec3c80c..5e1cfa9d8f 100644 --- a/slatedb/src/garbage_collector/manifest_gc.rs +++ b/slatedb/src/garbage_collector/manifest_gc.rs @@ -5,6 +5,7 @@ use chrono::{DateTime, Utc}; use futures::StreamExt; use log::error; use std::collections::HashSet; +use std::sync::atomic::{AtomicU64, Ordering}; use std::sync::Arc; use super::filter::retain_allowed_by_gc_filter; @@ -52,7 +53,9 @@ impl ManifestGcTask { /// Deletes the given manifests from the manifest store. /// /// In case of dryrun, the actual deletion doesn't happen. - async fn maybe_delete_manifests(&self, manifest_ids: Vec) { + /// + /// Returns the number of manifests actually deleted. + async fn maybe_delete_manifests(&self, manifest_ids: Vec) -> u64 { if self.manifest_options.dry_run { if !manifest_ids.is_empty() { log::info!( @@ -63,18 +66,24 @@ impl ManifestGcTask { for id in manifest_ids { log::debug!("dry run: would delete manifest but skipped [id={:?}]", id); } - return; + return 0; } + let deleted_count = AtomicU64::new(0); futures::stream::iter(manifest_ids) - .for_each_concurrent(GC_DELETE_CONCURRENCY, |id| async move { - if let Err(e) = self.manifest_store.delete_manifest_unchecked(id).await { - error!("error deleting manifest [id={:?}, error={}]", id, e); - } else { - self.stats.gc_manifest_count.increment(1); + .for_each_concurrent(GC_DELETE_CONCURRENCY, |id| { + let deleted_count = &deleted_count; + async move { + if let Err(e) = self.manifest_store.delete_manifest_unchecked(id).await { + error!("error deleting manifest [id={:?}, error={}]", id, e); + } else { + self.stats.gc_manifest_count.increment(1); + deleted_count.fetch_add(1, Ordering::Relaxed); + } } }) .await; + deleted_count.load(Ordering::Relaxed) } } @@ -103,6 +112,8 @@ impl GcTask for ManifestGcTask { .collect(); // Delete manifests older than min_age + // Capture length before into_iter() consumes the list; +1 re-adds the popped latest. + let pre_gc_count = manifest_metadata_list.len() as u64 + 1; let manifests_to_delete = manifest_metadata_list .into_iter() .filter(|manifest_metadata| { @@ -131,7 +142,15 @@ impl GcTask for ManifestGcTask { .map(|manifest_metadata| manifest_metadata.id) .collect::>(); - self.maybe_delete_manifests(manifest_ids_to_delete).await; + self.stats.gc_manifest_versions.set(pre_gc_count as i64); + + let deleted_count = self.maybe_delete_manifests(manifest_ids_to_delete).await; + + if deleted_count > 0 { + self.stats + .gc_manifest_versions + .set((pre_gc_count - deleted_count) as i64); + } Ok(()) } @@ -152,7 +171,9 @@ mod tests { use chrono::TimeDelta; use object_store::{memory::InMemory, path::Path, ObjectStoreExt}; use slatedb_common::clock::DefaultSystemClock; - use slatedb_common::metrics::MetricsRecorderHelper; + use slatedb_common::metrics::{ + lookup_metric_with_labels, DefaultMetricsRecorder, MetricsRecorderHelper, + }; use slatedb_common::ObjectMetadata; use std::time::Duration; @@ -329,4 +350,150 @@ mod tests { assert!(manifest_store.try_read_manifest(1).await.unwrap().is_some()); assert!(manifest_store.try_read_manifest(2).await.unwrap().is_some()); } + + async fn make_manifest_store() -> (Arc, StoredManifest) { + let object_store = Arc::new(InMemory::new()); + let manifest_store = Arc::new(ManifestStore::new(&Path::from("/root"), object_store)); + let stored = StoredManifest::create_new_db( + manifest_store.clone(), + ManifestCore::new(), + Arc::new(DefaultSystemClock::new()), + ) + .await + .unwrap(); + (manifest_store, stored) + } + + #[tokio::test] + async fn test_version_count_after_gc_deletes_old_manifests() { + let (manifest_store, mut stored) = make_manifest_store().await; + // Write two more manifests: ids 1 (create), 2, 3 + stored + .update(stored.prepare_dirty().unwrap()) + .await + .unwrap(); + stored + .update(stored.prepare_dirty().unwrap()) + .await + .unwrap(); + + let metrics = Arc::new(DefaultMetricsRecorder::new()); + let recorder = MetricsRecorderHelper::new(metrics.clone(), Default::default()); + let task = ManifestGcTask::new( + manifest_store.clone(), + Arc::new(GcStats::new(&recorder)), + GarbageCollectorDirectoryOptions { + min_age: Duration::ZERO, + interval: None, + dry_run: false, + }, + None, + true, + ); + + task.collect(Utc::now() + TimeDelta::hours(1)) + .await + .unwrap(); + + // GC deletes manifests 1 and 2 (older than min_age=0); manifest 3 survives as latest. + assert_eq!( + lookup_metric_with_labels( + &metrics, + crate::garbage_collector::stats::VERSION_COUNT, + &[("resource", "manifest")] + ), + Some(1), + "expected 1 surviving manifest after GC" + ); + } + + #[tokio::test] + async fn test_version_count_when_nothing_to_delete() { + let (manifest_store, mut stored) = make_manifest_store().await; + stored + .update(stored.prepare_dirty().unwrap()) + .await + .unwrap(); + stored + .update(stored.prepare_dirty().unwrap()) + .await + .unwrap(); + + let metrics = Arc::new(DefaultMetricsRecorder::new()); + let recorder = MetricsRecorderHelper::new(metrics.clone(), Default::default()); + let task = ManifestGcTask::new( + manifest_store.clone(), + Arc::new(GcStats::new(&recorder)), + GarbageCollectorDirectoryOptions { + min_age: Duration::from_secs(3600), // too new to delete + interval: None, + dry_run: false, + }, + None, + true, + ); + + task.collect(Utc::now()).await.unwrap(); + + // Nothing deleted — all 3 manifests survive. + assert_eq!( + lookup_metric_with_labels( + &metrics, + crate::garbage_collector::stats::VERSION_COUNT, + &[("resource", "manifest")] + ), + Some(3), + "expected all 3 manifests when nothing qualifies for deletion" + ); + } + + #[tokio::test] + async fn test_version_count_unchanged_on_dry_run() { + let (manifest_store, mut stored) = make_manifest_store().await; + // Write two more manifests: ids 1 (create), 2, 3 + stored + .update(stored.prepare_dirty().unwrap()) + .await + .unwrap(); + stored + .update(stored.prepare_dirty().unwrap()) + .await + .unwrap(); + + let metrics = Arc::new(DefaultMetricsRecorder::new()); + let recorder = MetricsRecorderHelper::new(metrics.clone(), Default::default()); + let task = ManifestGcTask::new( + manifest_store.clone(), + Arc::new(GcStats::new(&recorder)), + GarbageCollectorDirectoryOptions { + min_age: Duration::ZERO, + interval: None, + dry_run: true, + }, + None, + true, + ); + + task.collect(Utc::now() + TimeDelta::hours(1)) + .await + .unwrap(); + + // Dry run deletes nothing, so all 3 manifests still exist and the gauge reports + // the true current count rather than the hypothetical post-deletion count. + assert_eq!( + lookup_metric_with_labels( + &metrics, + crate::garbage_collector::stats::VERSION_COUNT, + &[("resource", "manifest")] + ), + Some(3), + "expected dry run to leave the gauge at the true current count" + ); + let manifests = manifest_store.list_manifests(..).await.unwrap(); + assert_eq!( + manifests.len(), + 3, + "dry run should not delete any manifests" + ); + } } diff --git a/slatedb/src/garbage_collector/stats.rs b/slatedb/src/garbage_collector/stats.rs index 66fe9ec8f4..c1b25f5702 100644 --- a/slatedb/src/garbage_collector/stats.rs +++ b/slatedb/src/garbage_collector/stats.rs @@ -1,4 +1,4 @@ -use slatedb_common::metrics::{CounterFn, MetricsRecorderHelper}; +use slatedb_common::metrics::{CounterFn, GaugeFn, MetricsRecorderHelper}; use std::sync::Arc; macro_rules! gc_stat_name { @@ -9,6 +9,7 @@ macro_rules! gc_stat_name { pub const DELETED_COUNT: &str = gc_stat_name!("deleted_count"); pub const GC_COUNT: &str = gc_stat_name!("count"); +pub const VERSION_COUNT: &str = gc_stat_name!("version_count"); /// Stats for the garbage collector. pub struct GcStats { @@ -19,6 +20,8 @@ pub struct GcStats { pub gc_compactions_count: Arc, pub gc_detach_count: Arc, pub gc_count: Arc, + pub gc_manifest_versions: Arc, + pub gc_compactions_versions: Arc, } impl GcStats { @@ -49,6 +52,14 @@ impl GcStats { .labels(&[("resource", "detach")]) .register(), gc_count: recorder.counter(GC_COUNT).register(), + gc_manifest_versions: recorder + .gauge(VERSION_COUNT) + .labels(&[("resource", "manifest")]) + .register(), + gc_compactions_versions: recorder + .gauge(VERSION_COUNT) + .labels(&[("resource", "compactions")]) + .register(), } } } diff --git a/slatedb/src/garbage_collector/wal_gc.rs b/slatedb/src/garbage_collector/wal_gc.rs index f4b06c2f54..ad6d2b5181 100644 --- a/slatedb/src/garbage_collector/wal_gc.rs +++ b/slatedb/src/garbage_collector/wal_gc.rs @@ -1,51 +1,29 @@ +use super::GcTask; use crate::manifest::Manifest; use crate::{ - config::GarbageCollectorDirectoryOptions, db_state::SsTableId, error::SlateDBError, - manifest::store::ManifestStore, tablestore::TableStore, + error::SlateDBError, + manifest::store::ManifestStore, + wal::{WalFileRange, WalGc}, }; use chrono::{DateTime, Utc}; -use futures::StreamExt; -use log::error; use std::collections::BTreeMap; +use std::ops::Bound; use std::sync::Arc; - -use super::filter::retain_allowed_by_gc_filter; -use super::{GcFilter, GcStats, GcTask, GC_DELETE_CONCURRENCY}; -use slatedb_common::object_metadata::IdentifiedObjectMetadata; - -/// Selects which class of WAL object a [`WalGcTask`] collects. -/// -/// Regular WAL SSTs and zero-byte WAL fence objects share the same WAL -/// directory and `SsTableId::Wal` identifier space, but they have separate -/// retention policies. This mode keeps a single task implementation while -/// allowing regular WAL GC and fence WAL GC to run on independent schedules. -#[derive(Debug, Clone, Copy)] -pub(super) enum WalGcMode { - /// Collect non-empty WAL SSTs that are older than the compacted WAL - /// boundary, old enough for retention, and unreferenced by active - /// checkpoint manifests. - Regular, - - /// Collect zero-byte WAL fence objects under the same safety checks as - /// regular WAL GC. - Fence, -} +use std::time::Duration; #[derive(Clone)] pub(crate) struct WalGcTask { manifest_store: Arc, - table_store: Arc, - stats: Arc, - wal_options: GarbageCollectorDirectoryOptions, - mode: WalGcMode, - gc_filter: Option>, + wal_gc: Arc, + resource: &'static str, + min_age: Duration, + dry_run: bool, } impl std::fmt::Debug for WalGcTask { fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { f.debug_struct("WalGcTask") - .field("wal_options", &self.wal_options) - .field("mode", &self.mode) + .field("resource", &self.resource.to_string()) .finish() } } @@ -53,136 +31,175 @@ impl std::fmt::Debug for WalGcTask { impl WalGcTask { pub(super) fn new( manifest_store: Arc, - table_store: Arc, - stats: Arc, - wal_options: GarbageCollectorDirectoryOptions, - mode: WalGcMode, - gc_filter: Option>, + wal_gc: Arc, + resource: &'static str, + min_age: Duration, + dry_run: bool, ) -> Self { - WalGcTask { + Self { manifest_store, - table_store, - stats, - wal_options, - mode, - gc_filter, + wal_gc, + resource, + min_age, + dry_run, } } - fn is_wal_sst_eligible_for_deletion( - utc_now: &DateTime, - wal_sst: &IdentifiedObjectMetadata, - min_age: &chrono::Duration, + fn referenced_wal_ranges( + latest_manifest_id: u64, active_manifests: &BTreeMap, - ) -> bool { - if utc_now.signed_duration_since(wal_sst.metadata.last_modified) <= *min_age { - return false; - } - - let wal_sst_id = wal_sst.id.unwrap_wal_id(); - !active_manifests - .values() - .any(|manifest| manifest.has_wal_sst_reference(wal_sst_id)) - } - - fn wal_sst_min_age(&self) -> chrono::Duration { - chrono::Duration::from_std(self.wal_options.min_age).expect("invalid duration") - } - - /// Deletes the given WAL SSTs from the table store. - /// - /// In case of dryrun, the actual deletion doesn't happen. - async fn maybe_delete_wal_ssts(&self, sst_ids: Vec) { - if self.wal_options.dry_run { - if !sst_ids.is_empty() { - log::info!( - "dry run: skipping {} deletion [count={}]", - self.resource(), - sst_ids.len() - ); - if matches!(self.mode, WalGcMode::Fence) { - log::info!( - "WAL fence GC is dry-run by default. This is a conservative setting. \ - Set wal_fence_options.dry_run=false and use a conservative min_age to enable. \ - Silence this log with wal_fence_options=None. See #352 for details." - ); - } - } - for id in sst_ids { - log::debug!( - "dry run: would delete {} but skipped [id={:?}]", - self.resource(), - id - ); - } - return; - } - - futures::stream::iter(sst_ids) - .for_each_concurrent(GC_DELETE_CONCURRENCY, |id| async move { - if let Err(e) = self.table_store.delete_sst(&id).await { - error!("error deleting WAL SST [id={:?}, error={}]", id, e); + ) -> Vec { + active_manifests + .iter() + .map(|(manifest_id, manifest)| { + if *manifest_id == latest_manifest_id { + // Keep the current compaction boundary and everything after it. Retaining the + // boundary matches the existing GC protocol and protects concurrent writers. + WalFileRange( + Bound::Included(manifest.core.replay_after_wal_id), + Bound::Unbounded, + ) } else { - match self.mode { - WalGcMode::Regular => self.stats.gc_wal_count.increment(1), - WalGcMode::Fence => self.stats.gc_wal_fence_count.increment(1), - } + // A checkpoint only references WALs that must be replayed for its manifest. + WalFileRange( + Bound::Excluded(manifest.core.replay_after_wal_id), + Bound::Excluded(manifest.core.next_wal_sst_id), + ) } }) - .await; + .collect() } } impl GcTask for WalGcTask { - /// Collect garbage from the WAL SSTs. This will delete any WAL SSTs that meet - /// the following conditions: - /// - not referenced by an active checkpoint - /// - older than the minimum age specified in the options - /// - older than the last compacted WAL SST. - async fn collect(&self, utc_now: DateTime) -> Result<(), SlateDBError> { + /// Resolve the WAL ranges referenced by the current manifest and active checkpoints, then + /// delegate collection to the configured WAL implementation. + async fn collect(&self, _utc_now: DateTime) -> Result<(), SlateDBError> { let latest_manifest = self.manifest_store.read_latest_manifest().await?; let active_manifests = self .manifest_store .read_referenced_manifests(latest_manifest.id, &latest_manifest.manifest) .await?; - let last_compacted_wal_sst_id = latest_manifest.manifest.core.replay_after_wal_id; - let min_age = self.wal_sst_min_age(); - let ssts_to_delete = self - .table_store - .list_wal_ssts(..last_compacted_wal_sst_id) - .await? - .into_iter() - .filter(|wal_sst| match self.mode { - // In regular mode, only consider WAL SSTs with size > 0 for deletion. - WalGcMode::Regular => wal_sst.metadata.size > 0, - // In fence mode, only consider zero-byte WAL SSTs for deletion. - WalGcMode::Fence => wal_sst.metadata.size == 0, - }) - // Respect min_age and any WAL references held by active checkpoint manifests. - .filter(|wal_sst| { - Self::is_wal_sst_eligible_for_deletion( - &utc_now, - wal_sst, - &min_age, - &active_manifests, - ) - }) - .collect::>(); - let ssts_to_delete = retain_allowed_by_gc_filter(&self.gc_filter, ssts_to_delete).await; - let sst_ids_to_delete = ssts_to_delete - .into_iter() - .map(|wal_sst| wal_sst.id) - .collect::>(); - - self.maybe_delete_wal_ssts(sst_ids_to_delete).await; + let referenced_ranges = Self::referenced_wal_ranges(latest_manifest.id, &active_manifests); - Ok(()) + self.wal_gc + .collect(referenced_ranges, self.min_age, self.dry_run) + .await + .map_err(Into::into) } fn resource(&self) -> &str { - match self.mode { - WalGcMode::Regular => "WAL", - WalGcMode::Fence => "WAL fence", + self.resource + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::checkpoint::Checkpoint; + use crate::manifest::store::StoredManifest; + use crate::manifest::ManifestCore; + use crate::wal::WalError; + use async_trait::async_trait; + use object_store::memory::InMemory; + use object_store::path::Path; + use object_store::ObjectStore; + use slatedb_common::clock::DefaultSystemClock; + use std::sync::Mutex; + use std::time::Duration; + use uuid::Uuid; + + #[derive(Default)] + struct RecordingWalGc { + calls: Mutex>>, + } + + impl RecordingWalGc { + fn calls(&self) -> Vec> { + self.calls.lock().unwrap().clone() + } + } + + #[async_trait] + impl WalGc for RecordingWalGc { + async fn collect( + &self, + referenced_ranges: Vec, + _min_age: Duration, + _dry_run: bool, + ) -> Result<(), WalError> { + self.calls.lock().unwrap().push(referenced_ranges); + Ok(()) } } + + #[test] + fn test_referenced_wal_ranges() { + let mut checkpoint_core = ManifestCore::new(); + checkpoint_core.replay_after_wal_id = 2; + checkpoint_core.next_wal_sst_id = 6; + + let mut current_core = ManifestCore::new(); + current_core.replay_after_wal_id = 5; + current_core.next_wal_sst_id = 8; + + let active_manifests = BTreeMap::from([ + (1, Manifest::initial(checkpoint_core)), + (2, Manifest::initial(current_core)), + ]); + + assert_eq!( + WalGcTask::referenced_wal_ranges(2, &active_manifests), + vec![ + WalFileRange(Bound::Excluded(2), Bound::Excluded(6)), + WalFileRange(Bound::Included(5), Bound::Unbounded), + ] + ); + } + + #[tokio::test] + async fn test_collect_calls_wal_gc_with_referenced_ranges() { + let object_store: Arc = Arc::new(InMemory::new()); + let manifest_store = Arc::new(ManifestStore::new( + &Path::from("/test/wal-gc-ranges"), + object_store, + )); + + let mut checkpoint_core = ManifestCore::new(); + checkpoint_core.replay_after_wal_id = 2; + checkpoint_core.next_wal_sst_id = 6; + let mut stored_manifest = StoredManifest::create_new_db( + manifest_store.clone(), + checkpoint_core, + Arc::new(DefaultSystemClock::new()), + ) + .await + .unwrap(); + let checkpoint_manifest_id = stored_manifest.id(); + + let mut dirty = stored_manifest.prepare_dirty().unwrap(); + dirty.value.core.replay_after_wal_id = 5; + dirty.value.core.next_wal_sst_id = 8; + dirty.value.core.checkpoints.push(Checkpoint { + id: Uuid::new_v4(), + manifest_id: checkpoint_manifest_id, + expire_time: None, + create_time: Utc::now(), + name: None, + }); + stored_manifest.update(dirty).await.unwrap(); + + let wal_gc = Arc::new(RecordingWalGc::default()); + let task = WalGcTask::new(manifest_store, wal_gc.clone(), "WAL", Duration::ZERO, false); + + task.collect(Utc::now()).await.unwrap(); + + assert_eq!( + wal_gc.calls(), + vec![vec![ + WalFileRange(Bound::Excluded(2), Bound::Excluded(6)), + WalFileRange(Bound::Included(5), Bound::Unbounded), + ]] + ); + } } diff --git a/slatedb/src/lib.rs b/slatedb/src/lib.rs index a7767bba15..cdd1e69aa9 100644 --- a/slatedb/src/lib.rs +++ b/slatedb/src/lib.rs @@ -76,12 +76,12 @@ pub use query_metrics::{ pub use slatedb_common::{DbRand, IdentifiedObjectMetadata, ObjectMetadata}; #[cfg(test)] pub use sst_builder::BlockFormat; -pub use sst_reader::{SstFile, SstReader}; +pub use sst_reader::{SstFile, SstIndex, SstReader}; pub use sst_stats::{BlockStats, SstStats}; pub use transaction_manager::IsolationLevel; pub use types::KeyValue; pub use types::{RowEntry, ValueDeletable}; -pub use wal_buffer::stats as wal_buffer_stats; +pub use wal::slatedb::writer::stats as wal_buffer_stats; pub use wal_reader::{WalFile, WalFileIterator, WalReader}; pub mod admin; @@ -168,6 +168,7 @@ mod single_flight; mod snapshot_manager; mod sorted_run_iterator; mod sst_builder; +mod sst_io; mod sst_iter; mod sst_reader; mod sst_stats; @@ -180,7 +181,6 @@ mod types; mod utils; mod fence; -mod wal_buffer; mod wal_reader; mod wal_replay; @@ -190,5 +190,5 @@ mod wal_replay; #[cfg(test)] #[ctor::ctor] fn init_test_infrastructure() { - crate::test_utils::init_test_infrastructure(); + test_utils::init_test_infrastructure(); } diff --git a/slatedb/src/manifest/mod.rs b/slatedb/src/manifest/mod.rs index 1995d9e334..2558f39f13 100644 --- a/slatedb/src/manifest/mod.rs +++ b/slatedb/src/manifest/mod.rs @@ -22,6 +22,10 @@ pub(crate) mod store; pub use crate::db_state::{SortedRun, SsTableHandle, SsTableId, SsTableInfo, SsTableView}; +/// The per-source trees a union merges into each segment, keyed by segment +/// prefix and ordered within a segment by that segment's start key. +type SegmentContributors<'a> = BTreeMap>; + /// Per-LSM-tree state. Shared shape between the unsegmented tree (held directly /// on `ManifestCore`) and each named segment held in `ManifestCore::segments`. #[derive(Clone, Default, PartialEq, Serialize, Debug)] @@ -66,6 +70,14 @@ impl LsmTreeState { self.l0.is_empty() && self.compacted.is_empty() } + /// Iterate every SST view referenced by this tree — L0 views followed by + /// the views of every sorted run. + pub(crate) fn sst_views(&self) -> impl Iterator { + self.l0 + .iter() + .chain(self.compacted.iter().flat_map(|sr| sr.sst_views().iter())) + } + /// Total number of SST views referenced by this tree — L0 plus every /// SST in every sorted run. Used by the read path to size scan /// parallelism. @@ -74,7 +86,7 @@ impl LsmTreeState { + self .compacted .iter() - .map(|sr| sr.sst_views.len()) + .map(|sr| sr.sst_views().len()) .sum::() } @@ -551,11 +563,7 @@ impl ManifestCore { /// Iterate every SST view referenced by this manifest — L0 views and /// sorted-run views across the unsegmented tree and every segment. pub(crate) fn all_sst_views(&self) -> impl Iterator { - self.trees().flat_map(|tree| { - tree.l0 - .iter() - .chain(tree.compacted.iter().flat_map(|sr| sr.sst_views.iter())) - }) + self.trees().flat_map(|tree| tree.sst_views()) } /// Compare a configured WAL-store URI against the manifest's @@ -1151,12 +1159,9 @@ impl Manifest { let l0: VecDeque = Self::filter_view_handles(&tree.l0, true, range).into(); let mut sorted_runs_filtered = vec![]; for sr in &tree.compacted { - let sst_views = Self::filter_view_handles(&sr.sst_views, false, range); + let sst_views = Self::filter_view_handles(sr.sst_views().iter(), false, range); if !sst_views.is_empty() { - sorted_runs_filtered.push(SortedRun { - id: sr.id, - sst_views, - }); + sorted_runs_filtered.push(SortedRun::new(sr.id, sst_views)); } } tree.l0 = l0; @@ -1293,38 +1298,90 @@ impl Manifest { Ok(()) } - /// Extractor-configured case. Build a per-prefix accumulator from - /// every source's `core.segments`, validate the antichain, and write - /// the result into `core.segments`. Empty entries (no L0, no compacted - /// runs) are dropped — the unioned manifest should not carry - /// placeholders. Watermarks are intentionally not carried over: the - /// unioned manifest is a fresh DB that begins compaction tracking from - /// scratch. - fn build_segmented_lsm_state( - core: &mut ManifestCore, - sources: &[&CloneSource], - ) -> Result<(), SlateDBError> { - let mut segments_by_prefix: BTreeMap = BTreeMap::new(); + /// Extractor-configured case. Concatenate the per-source trees that + /// [`Self::group_segments_for_union`] collected for each prefix into one + /// tree per segment. Segments holding no SSTs are already absent from + /// `segments`, so the unioned manifest carries no placeholders. + /// Watermarks are intentionally not carried over: the unioned manifest is + /// a fresh DB that begins compaction tracking from scratch. + fn build_segmented_lsm_state(core: &mut ManifestCore, segments: SegmentContributors) { + core.segments = segments + .into_iter() + .map(|(prefix, trees)| { + let mut merged = LsmTreeState::default(); + for tree in trees { + merged.l0.extend(tree.l0.iter().cloned()); + merged.compacted.extend(tree.compacted.iter().cloned()); + } + Segment { + prefix, + tree: Arc::new(merged), + } + }) + .collect(); + } + + /// Collect the trees that each source contributes to each segment. Within + /// a segment the trees are in key order, so the merged segment's `l0` and + /// `compacted` lists come out sorted — the same thing the manifest-wide + /// sort does for the unsegmented tree. + /// + /// Two things are rejected. If one segment prefix is a prefix of another, + /// the two segments cover some of the same keys. If two sources hold the + /// same keys inside one segment, the merged segment cannot say which + /// source owns them. + /// + /// A segment with no SSTs contributes nothing, so it is left out. + fn group_segments_for_union<'a>( + sources: &[&'a CloneSource], + ) -> Result, SlateDBError> { + let all_prefixes: BTreeSet<&Bytes> = sources + .iter() + .flat_map(|source| source.manifest.core.segments.iter().map(|seg| &seg.prefix)) + .collect(); + Self::ensure_union_prefix_antichain(all_prefixes)?; + + let mut by_prefix: BTreeMap> = BTreeMap::new(); for source in sources { for segment in &source.manifest.core.segments { - let entry = segments_by_prefix - .entry(segment.prefix.clone()) - .or_default(); - entry.l0.extend(segment.tree.l0.iter().cloned()); - entry - .compacted - .extend(segment.tree.compacted.iter().cloned()); + if let Some(range) = Self::bounding_range(segment.tree.sst_views()) { + by_prefix + .entry(segment.prefix.clone()) + .or_default() + .push((range, segment.tree.as_ref())); + } + } + } + + let mut grouped = SegmentContributors::new(); + for (prefix, mut members) in by_prefix { + members.sort_by_key(|(range, _)| range.comparable_start_bound().cloned()); + let ranges: Vec = members.iter().map(|(range, _)| range.clone()).collect(); + Self::ensure_disjoint_ranges(Some(&prefix), &ranges)?; + grouped.insert(prefix, members.into_iter().map(|(_, tree)| tree).collect()); + } + Ok(grouped) + } + + /// Reject overlapping source ranges. `ranges` must be sorted by start + /// bound. `segment` names the segment under check and is omitted from the + /// message for the unsegmented tree. + fn ensure_disjoint_ranges( + segment: Option<&Bytes>, + ranges: &[BytesRange], + ) -> Result<(), SlateDBError> { + for pair in ranges.windows(2) { + if pair[1].intersect(&pair[0]).is_some() { + let scope = match segment { + Some(prefix) => format!(" in segment `{:?}`", prefix), + None => String::new(), + }; + return Err(SlateDBError::InvalidUnion(format!( + "clone sources have overlapping key ranges{}. ranges=`{:?}`", + scope, ranges + ))); } } - Self::ensure_union_prefix_antichain(segments_by_prefix.keys())?; - core.segments = segments_by_prefix - .into_iter() - .filter(|(_, tree)| !tree.is_empty()) - .map(|(prefix, tree)| Segment { - prefix, - tree: Arc::new(tree), - }) - .collect(); Ok(()) } @@ -1404,30 +1461,17 @@ impl Manifest { } ranges.sort_by_key(|(_, range)| range.comparable_start_bound().cloned()); - // Ensure source key ranges are non-overlapping. Surfaces as a typed - // error since the source set is user-supplied. - let mut previous_range = None; - for (_, range) in ranges.iter() { - if let Some(previous_range) = previous_range { - if range.intersect(previous_range).is_some() { - let all: Vec = ranges.iter().map(|(_, r)| (*r).clone()).collect(); - return Err(SlateDBError::InvalidUnion(format!( - "clone sources have overlapping key ranges. ranges=`{:?}`", - all - ))); - } - } - previous_range = Some(range); - } - let ordered_sources: Vec<&CloneSource> = ranges.iter().map(|(s, _)| *s).collect(); let mut core = ManifestCore::new(); core.segment_extractor_name = Self::ensure_consistent_segment_extractor(&sources)?; if core.segment_extractor_name.is_none() { + let all: Vec = ranges.iter().map(|(_, r)| (*r).clone()).collect(); + Self::ensure_disjoint_ranges(None, &all)?; Self::build_unsegmented_lsm_state(&mut core, &ordered_sources)?; } else { - Self::build_segmented_lsm_state(&mut core, &ordered_sources)?; + let segments = Self::group_segments_for_union(&ordered_sources)?; + Self::build_segmented_lsm_state(&mut core, segments); } Self::renumber_union_sorted_runs(&mut core); @@ -1488,9 +1532,15 @@ impl Manifest { } fn range(&self) -> Option { + Self::bounding_range(self.core.all_sst_views()) + } + + /// Smallest range covering every view in `views`, or `None` when `views` + /// is empty. + fn bounding_range<'a>(views: impl Iterator) -> Option { let mut start_bound = None; let mut end_bound = None; - for sst in self.core.all_sst_views() { + for sst in views { let range = sst.compacted_effective_range(); start_bound = start_bound .map(|b| min(b, range.comparable_start_bound())) @@ -1546,10 +1596,6 @@ impl Manifest { .collect() } - pub(crate) fn has_wal_sst_reference(&self, wal_sst_id: u64) -> bool { - wal_sst_id > self.core.replay_after_wal_id && wal_sst_id < self.core.next_wal_sst_id - } - /// Shrinks each `ExternalDb.sst_ids` to only IDs still referenced by this manifest's /// L0 and compacted sorted runs. `ExternalDb` entries are retained even when their /// `sst_ids` becomes empty — detaching a clone from its parent is done by the GC, @@ -1606,7 +1652,7 @@ mod tests { .await .unwrap(); let checkpoint = parent_manifest - .write_checkpoint(uuid::Uuid::new_v4(), &CheckpointOptions::default()) + .write_checkpoint(Uuid::new_v4(), &CheckpointOptions::default()) .await .unwrap(); @@ -1748,7 +1794,7 @@ mod tests { .unwrap(); let checkpoint = manifest - .write_checkpoint(uuid::Uuid::new_v4(), &CheckpointOptions::default()) + .write_checkpoint(Uuid::new_v4(), &CheckpointOptions::default()) .await .unwrap(); @@ -1997,7 +2043,7 @@ mod tests { fn test_union(#[case] test_case: UnionTestCase) { let mut sst_ids: HashMap = HashMap::new(); let rand = Arc::new(DbRand::default()); - let sources: Vec = test_case + let sources: Vec = test_case .manifests .iter() .enumerate() @@ -2086,7 +2132,7 @@ mod tests { l0: writer_l0.clone(), compacted: vec![], }; - let compactor_compacted = vec![SortedRun { id: 42, sst_views: vec![] }]; + let compactor_compacted = vec![SortedRun::new(42, [])]; let compactor = LsmTreeState { last_compacted_l0_sst_view_id: last_view, last_compacted_l0_sst_id: last_sst, @@ -2469,13 +2515,8 @@ mod tests { } tree.last_compacted_l0_sst_view_id = Some(newest); self.next_sr_id += 1; - tree.compacted.insert( - 0, - SortedRun { - id: self.next_sr_id, - sst_views: sr_views, - }, - ); + tree.compacted + .insert(0, SortedRun::new(self.next_sr_id, sr_views)); } } @@ -2574,7 +2615,7 @@ mod tests { } for seg in &self.store { for sr in &seg.tree.compacted { - for view in &sr.sst_views { + for view in sr.sst_views() { assert!( self.flushed_l0s.contains(&view.id), "SR {} references L0 {} that was never flushed", @@ -2661,7 +2702,7 @@ mod tests { for seg in &self.store { let l0_ids: BTreeSet = seg.tree.l0.iter().map(|v| v.id).collect(); for sr in &seg.tree.compacted { - for view in &sr.sst_views { + for view in sr.sst_views() { assert!( !l0_ids.contains(&view.id), "L0 {} appears in both l0 list and SR {} of segment {:?}", @@ -2767,7 +2808,7 @@ mod tests { ); // Verify no duplicates - let mut seen = std::collections::HashSet::new(); + let mut seen = HashSet::new(); for id in &sr_ids { assert!(seen.insert(id), "Duplicate SR ID: {}", id); } @@ -2978,28 +3019,25 @@ mod tests { )); } for (idx, sorted_run) in manifest.sorted_runs.iter().enumerate() { - Arc::make_mut(&mut core.tree).compacted.push(SortedRun { - id: idx as u32, - sst_views: sorted_run - .iter() - .map(|entry| { - let sst_id = sst_id_fn(entry.sst_alias); - let view_id = sst_id.unwrap_compacted_id(); - SsTableView::new_projected( - view_id, - SsTableHandle::new( - sst_id, - SST_FORMAT_VERSION_LATEST, - SsTableInfo { - first_entry: Some(entry.first_entry.clone()), - ..SsTableInfo::default() - }, - ), - entry.visible_range.clone(), - ) - }) - .collect(), - }); + Arc::make_mut(&mut core.tree).compacted.push(SortedRun::new( + idx as u32, + sorted_run.iter().map(|entry| { + let sst_id = sst_id_fn(entry.sst_alias); + let view_id = sst_id.unwrap_compacted_id(); + SsTableView::new_projected( + view_id, + SsTableHandle::new( + sst_id, + SST_FORMAT_VERSION_LATEST, + SsTableInfo { + first_entry: Some(entry.first_entry.clone()), + ..SsTableInfo::default() + }, + ), + entry.visible_range.clone(), + ) + }), + )); } Manifest::initial(core) } @@ -3172,10 +3210,9 @@ mod tests { Arc::make_mut(&mut core.tree) .l0 .push_back(create_sst_view(live_l0, b"a")); - Arc::make_mut(&mut core.tree).compacted.push(SortedRun { - id: 0, - sst_views: vec![create_sst_view(live_compacted, b"b")], - }); + Arc::make_mut(&mut core.tree) + .compacted + .push(SortedRun::new(0, [create_sst_view(live_compacted, b"b")])); let mut manifest = Manifest::initial(core); manifest.external_dbs = vec![ @@ -3229,10 +3266,10 @@ mod tests { last_compacted_l0_sst_view_id: None, last_compacted_l0_sst_id: None, l0: VecDeque::from(vec![create_sst_view(segment_l0, b"seg/a")]), - compacted: vec![SortedRun { - id: 0, - sst_views: vec![create_sst_view(segment_compacted, b"seg/b")], - }], + compacted: vec![SortedRun::new( + 0, + [create_sst_view(segment_compacted, b"seg/b")], + )], }), }]; @@ -3276,9 +3313,9 @@ mod tests { visible_range: BytesRange, ) -> Manifest { let mut core = ManifestCore::new(); - Arc::make_mut(&mut core.tree).compacted.push(SortedRun { - id: 0, - sst_views: vec![SsTableView::new_projected( + Arc::make_mut(&mut core.tree).compacted.push(SortedRun::new( + 0, + [SsTableView::new_projected( sst_id.unwrap_compacted_id(), SsTableHandle::new( sst_id, @@ -3290,7 +3327,7 @@ mod tests { ), Some(visible_range), )], - }); + )); Manifest::initial(core) } @@ -3718,14 +3755,14 @@ mod tests { ); } - fn segment_with_prefix(prefix: &[u8], seed: u64) -> super::Segment { + fn segment_with_prefix(prefix: &[u8], seed: u64) -> Segment { let view_id = Ulid::from_parts(seed, 0); let handle = SsTableHandle::new( SsTableId::Compacted(Ulid::from_parts(seed, 1)), SST_FORMAT_VERSION_LATEST, SsTableInfo::default(), ); - super::Segment { + Segment { prefix: Bytes::copy_from_slice(prefix), tree: Arc::new(LsmTreeState { last_compacted_l0_sst_view_id: None, @@ -3736,7 +3773,7 @@ mod tests { } } - fn collect_prefixes(segments: &[super::Segment]) -> Vec { + fn collect_prefixes(segments: &[Segment]) -> Vec { segments.iter().map(|s| s.prefix.clone()).collect() } @@ -3975,9 +4012,9 @@ mod tests { ), Some(visible_range.clone()), )]), - compacted: vec![SortedRun { - id: 0, // gets renumbered globally by the union - sst_views: vec![SsTableView::new_projected( + compacted: vec![SortedRun::new( + 0, // gets renumbered globally by the union + [SsTableView::new_projected( sr_sst.unwrap_compacted_id(), SsTableHandle::new( sr_sst, @@ -3989,7 +4026,7 @@ mod tests { ), Some(visible_range), )], - }], + )], }), }]; (Manifest::initial(core), l0_sst, sr_sst) @@ -4047,10 +4084,7 @@ mod tests { // entry has the highest id (matching the descending-id-by-list- // position convention). fn make_sr(id: u32) -> SortedRun { - SortedRun { - id, // intentionally collides across trees pre-renumber - sst_views: vec![], - } + SortedRun::new(id, []) // intentionally collides across trees pre-renumber } let mut core = ManifestCore::new(); @@ -4252,9 +4286,9 @@ mod tests { ), Some(range.clone()), )]), - compacted: vec![SortedRun { - id: 0, - sst_views: vec![SsTableView::new_projected( + compacted: vec![SortedRun::new( + 0, + [SsTableView::new_projected( sr_id.unwrap_compacted_id(), SsTableHandle::new( sr_id, @@ -4266,7 +4300,7 @@ mod tests { ), Some(range), )], - }], + )], }), } } @@ -4325,6 +4359,146 @@ mod tests { assert_ne!(sr_ids[0], sr_ids[1]); } + /// Build a segmented manifest whose segments each hold a single L0 view + /// over the given range. `segments` is `(prefix, first_entry, range)` and + /// must be ordered by prefix, as `ManifestCore::segments` is. + fn segmented_shard_manifest( + extractor_name: &str, + segments: Vec<(&'static [u8], &'static str, BytesRange)>, + ) -> Manifest { + let mut core = ManifestCore::new(); + core.segment_extractor_name = Some(extractor_name.to_string()); + core.segments = segments + .into_iter() + .map(|(prefix, first_entry, range)| { + let sst = SsTableId::Compacted(Ulid::new()); + Segment { + prefix: Bytes::from_static(prefix), + tree: Arc::new(LsmTreeState { + last_compacted_l0_sst_view_id: None, + last_compacted_l0_sst_id: None, + l0: VecDeque::from(vec![SsTableView::new_projected( + sst.unwrap_compacted_id(), + SsTableHandle::new( + sst, + SST_FORMAT_VERSION_LATEST, + SsTableInfo { + first_entry: Some(Bytes::from_static(first_entry.as_bytes())), + ..SsTableInfo::default() + }, + ), + Some(range), + )]), + compacted: vec![], + }), + } + }) + .collect(); + Manifest::initial(core) + } + + fn union_of(manifests: Vec) -> Result { + Manifest::cloned_from_union( + manifests + .into_iter() + .enumerate() + .map(|(i, manifest)| CloneSource { + manifest, + path: Path::from(format!("/tmp/db{}", i)), + checkpoint: new_checkpoint(Uuid::new_v4()), + }) + .collect(), + Arc::new(DbRand::default()), + ) + } + + /// Effective range of each L0 view in the named segment, in list order. + fn segment_l0_ranges(manifest: &Manifest, prefix: &[u8]) -> Vec { + manifest + .core + .segments + .iter() + .find(|s| s.prefix == prefix) + .expect("segment present") + .tree + .l0 + .iter() + .map(|view| view.compacted_effective_range().clone()) + .collect() + } + + #[test] + fn test_union_validates_key_ranges_per_segment() { + // Two tenant shards of a store keyed `data/{tenant}` and + // `idx/{tenant}`. Each shard holds part of both segments, so the + // shards' bounding ranges overlap — `data/metro` sorts below + // `idx/bronx` — while neither segment does. A read routes to exactly + // one segment, so the union is unambiguous and must be accepted. + // + // The shards also lead in different segments: `left` holds the lower + // `data` keys, `right` the lower `idx` keys. Each segment's list must + // follow its own key order rather than the manifest-wide source + // order, so a source's entries stay contiguous and ascending within + // the segment, as they do in the unsegmented case. + fn shard(data: BytesRange, idx: BytesRange) -> Manifest { + segmented_shard_manifest("kind", vec![(b"data", "data", data), (b"idx", "idx", idx)]) + } + let left = shard( + BytesRange::from_ref("data/bronx".."data/metro"), + BytesRange::from_ref("idx/metro".."idx/zzz"), + ); + let right = shard( + BytesRange::from_ref("data/metro".."data/zzz"), + BytesRange::from_ref("idx/bronx".."idx/metro"), + ); + + // The precondition the old manifest-wide check enforced is genuinely + // violated here; only the per-segment check makes this union legal. + let left_range = left.range().expect("left range"); + assert!(left_range + .intersect(&right.range().expect("right range")) + .is_some()); + + let union = union_of(vec![left, right]).expect("union of disjoint segments"); + + assert_eq!(union.core.segment_extractor_name.as_deref(), Some("kind")); + assert_eq!( + segment_l0_ranges(&union, b"data"), + vec![ + BytesRange::from_ref("data/bronx".."data/metro"), + BytesRange::from_ref("data/metro".."data/zzz"), + ] + ); + assert_eq!( + segment_l0_ranges(&union, b"idx"), + vec![ + BytesRange::from_ref("idx/bronx".."idx/metro"), + BytesRange::from_ref("idx/metro".."idx/zzz"), + ] + ); + + // Overlap *within* a shared segment is still rejected: the merged + // segment's read chain could not say which source owns the key. + let overlapping = shard( + BytesRange::from_ref("data/lincoln".."data/zzz"), + BytesRange::from_ref("idx/bronx".."idx/metro"), + ); + let err = union_of(vec![ + shard( + BytesRange::from_ref("data/bronx".."data/metro"), + BytesRange::from_ref("idx/metro".."idx/zzz"), + ), + overlapping, + ]) + .expect_err("overlap within `data`"); + let SlateDBError::InvalidUnion(msg) = err else { + panic!("expected InvalidUnion, got {:?}", err); + }; + // The message names the offending segment, not just the ranges. + assert!(msg.contains("in segment"), "{}", msg); + assert!(msg.contains("data"), "{}", msg); + } + #[test] fn test_union_unsegmented_sources_land_in_core_tree() { // Two sources with no extractor configured. The unioned manifest @@ -4693,9 +4867,9 @@ mod tests { ), Some(BytesRange::from_ref("a".."d")), )]), - compacted: vec![SortedRun { - id: 0, - sst_views: vec![SsTableView::new_projected( + compacted: vec![SortedRun::new( + 0, + [SsTableView::new_projected( sr_a.unwrap_compacted_id(), SsTableHandle::new( sr_a, @@ -4707,7 +4881,7 @@ mod tests { ), Some(BytesRange::from_ref("a".."m")), )], - }], + )], }), }, Segment { @@ -4716,9 +4890,9 @@ mod tests { last_compacted_l0_sst_view_id: None, last_compacted_l0_sst_id: None, l0: VecDeque::new(), - compacted: vec![SortedRun { - id: 1, - sst_views: vec![SsTableView::new_projected( + compacted: vec![SortedRun::new( + 1, + [SsTableView::new_projected( sr_b.unwrap_compacted_id(), SsTableHandle::new( sr_b, @@ -4730,7 +4904,7 @@ mod tests { ), Some(BytesRange::from_ref("n".."z")), )], - }], + )], }), }, ]; @@ -4774,13 +4948,13 @@ mod tests { ) }; let mut core = ManifestCore::new(); - Arc::make_mut(&mut core.tree).compacted.push(SortedRun { - id: 0, - sst_views: vec![ + Arc::make_mut(&mut core.tree).compacted.push(SortedRun::new( + 0, + [ make(sst1, b"a", b"c", BytesRange::from_ref("a".."d")), make(sst2, b"m", b"p", BytesRange::from_ref("m".."q")), ], - }); + )); Manifest::initial(core) } @@ -4814,7 +4988,7 @@ mod tests { ) .unwrap(); assert_eq!(projected.core.tree.compacted.len(), 1); - assert_eq!(projected.core.tree.compacted[0].sst_views.len(), 1); + assert_eq!(projected.core.tree.compacted[0].sst_views().len(), 1); } #[test] diff --git a/slatedb/src/manifest/store.rs b/slatedb/src/manifest/store.rs index 2128df67cf..6d0eba473c 100644 --- a/slatedb/src/manifest/store.rs +++ b/slatedb/src/manifest/store.rs @@ -687,7 +687,6 @@ pub(crate) mod test_utils { mod tests { use crate::checkpoint::Checkpoint; use crate::config::CheckpointOptions; - use crate::error; use crate::error::SlateDBError; use crate::manifest::store::{FenceableManifest, ManifestStore, StoredManifest}; use crate::manifest::ManifestCore; @@ -725,7 +724,7 @@ mod tests { assert!(matches!( result.unwrap_err(), - error::SlateDBError::TransactionalObjectVersionExists + SlateDBError::TransactionalObjectVersionExists )); } @@ -900,7 +899,7 @@ mod tests { .unwrap(); let result = writer1.refresh().await; - assert!(matches!(result, Err(error::SlateDBError::Fenced))); + assert!(matches!(result, Err(SlateDBError::Fenced))); } #[tokio::test] @@ -952,7 +951,7 @@ mod tests { .unwrap(); let result = compactor1.refresh().await; - assert!(matches!(result, Err(error::SlateDBError::Fenced))); + assert!(matches!(result, Err(SlateDBError::Fenced))); } #[tokio::test] @@ -1004,7 +1003,7 @@ mod tests { .write_checkpoint(uuid::Uuid::new_v4(), &CheckpointOptions::default()) .await; - assert!(matches!(result, Err(error::SlateDBError::Fenced))); + assert!(matches!(result, Err(SlateDBError::Fenced))); assert_state_not_updated(&mut compactor2).await; } @@ -1197,7 +1196,7 @@ mod tests { assert_eq!(manifests[1].id, 2); let result = ms.delete_manifest(2).await; - assert!(matches!(result, Err(error::SlateDBError::InvalidDeletion))); + assert!(matches!(result, Err(SlateDBError::InvalidDeletion))); } fn new_memory_manifest_store() -> Arc { diff --git a/slatedb/src/mem_table.rs b/slatedb/src/mem_table.rs index 23cce96a1d..b132cf465f 100644 --- a/slatedb/src/mem_table.rs +++ b/slatedb/src/mem_table.rs @@ -212,6 +212,8 @@ impl RowEntryIterator for MemTableIterator { let front = self.borrow_item().clone(); if front.is_some_and(|record| record.key < next_key) { self.next_sync(); + // Keep in-memory seeking cooperative. + tokio::task::coop::consume_budget().await; } else { return Ok(()); } @@ -527,13 +529,12 @@ impl KVTable { // because the monotonicity is enforced when generating the clock tick // (see [crate::utils::MonotonicClock::now]) if let Some(create_ts) = row.create_ts { - self.last_tick - .fetch_max(create_ts, atomic::Ordering::SeqCst); + self.last_tick.fetch_max(create_ts, SeqCst); } // update the last seq number if it is greater than the current last seq - self.last_seq.fetch_max(row.seq, atomic::Ordering::SeqCst); + self.last_seq.fetch_max(row.seq, SeqCst); // update the first seq number if it is smaller than the current first seq - self.first_seq.fetch_min(row.seq, atomic::Ordering::SeqCst); + self.first_seq.fetch_min(row.seq, SeqCst); let row_size = row.estimated_size(); self.map.compare_insert(internal_key, row, |previous_row| { @@ -848,11 +849,9 @@ mod tests { let sample_table = sample::table(runner.rng(), 500, 10); let kv_table = WritableKVTable::new(); - let mut seq = 1; - for (key, value) in &sample_table { + for (seq, (key, value)) in (1..).zip(&sample_table) { let row_entry = RowEntry::new_value(key, value, seq); kv_table.put(row_entry); - seq += 1; } runner diff --git a/slatedb/src/memtable_flusher/manifest_writer.rs b/slatedb/src/memtable_flusher/manifest_writer.rs index a9e4055a98..f138c07a8b 100644 --- a/slatedb/src/memtable_flusher/manifest_writer.rs +++ b/slatedb/src/memtable_flusher/manifest_writer.rs @@ -19,10 +19,11 @@ use super::uploader::UploadedMemtable; use crate::checkpoint::CheckpointCreateResult; use crate::config::CheckpointOptions; use crate::db::DbInner; -use crate::db_state::{collect_touched_segments, COWDbState, DbState, SsTableId, SsTableView}; +use crate::db_state::{collect_touched_segments, DbState, SsTableId, SsTableView}; use crate::dispatcher::MessageHandler; use crate::error::SlateDBError; use crate::manifest::store::FenceableManifest; +use crate::manifest::Manifest; use crate::oracle::Oracle; use crate::utils::IdGenerator; use crate::utils::SafeSender; @@ -32,6 +33,7 @@ use bytes::Bytes; use futures::stream::BoxStream; use futures::StreamExt; use parking_lot::RwLockWriteGuard; +use slatedb_txn_obj::DirtyObject; use std::cmp; use std::collections::{BTreeMap, HashSet}; use std::sync::Arc; @@ -620,9 +622,7 @@ impl ManifestWriterHandler { Ok(()) } - fn clone_local_manifest_for_write( - &self, - ) -> slatedb_txn_obj::DirtyObject { + fn clone_local_manifest_for_write(&self) -> DirtyObject { let dirty = { let rguard_state = self.db.state.read(); rguard_state.state().manifest.clone() @@ -655,29 +655,23 @@ impl ManifestWriterHandler { result } - fn merge_remote_manifest( - &self, - remote_dirty: slatedb_txn_obj::DirtyObject, - ) { - let dirty_manifest = { + fn merge_remote_manifest(&self, remote_dirty: DirtyObject) { + let manifest = { let mut wguard_state = self.db.state.write(); wguard_state.merge_remote_manifest(remote_dirty); - let cow = wguard_state.state(); - self.update_stats_for_manifest(&cow); - cow.manifest.clone() + wguard_state.state().manifest.clone() }; - self.db - .status_manager - .report_manifest(dirty_manifest.into()); + self.update_stats_for_manifest(&manifest); + self.db.status_manager.report_manifest(manifest.into()); } - fn update_stats_for_manifest(&self, cow: &COWDbState) { + fn update_stats_for_manifest(&self, manifest: &DirtyObject) { let mut l0_ssts = 0usize; let mut segment_max_l0_ssts = 0usize; let mut sorted_runs = 0usize; let mut sst_views = 0usize; let mut distinct_ssts: HashSet = HashSet::new(); - for tree in cow.core().trees() { + for tree in manifest.value.core.trees() { l0_ssts += tree.l0.len(); // Track the largest single tree: backpressure is driven by `segment_max_l0_sst_count` // because `l0_max_ssts` is enforced per-tree. @@ -686,7 +680,7 @@ impl ManifestWriterHandler { let all_views = tree .l0 .iter() - .chain(tree.compacted.iter().flat_map(|run| run.sst_views.iter())); + .chain(tree.compacted.iter().flat_map(|run| run.sst_views().iter())); for view in all_views { sst_views += 1; // Dedupe by physical SST id: a range clone/rescale can project one SST into @@ -705,7 +699,7 @@ impl ManifestWriterHandler { self.db .db_stats .external_db_count - .set(cow.manifest.value.external_dbs.len() as i64); + .set(manifest.value.external_dbs.len() as i64); } async fn write_checkpoint_safely( @@ -737,10 +731,7 @@ impl ManifestWriterHandler { self.db.oracle.advance_durable_seq(uploaded.last_seq); } self.resolve_pending_flushes(); - for (checkpoint, result) in attached_checkpoints - .into_iter() - .zip(checkpoint_results.into_iter()) - { + for (checkpoint, result) in attached_checkpoints.into_iter().zip(checkpoint_results) { debug!("checkpoint created [id={}]", result.id); let _ = checkpoint.sender.send(Ok(result)); } @@ -1371,7 +1362,7 @@ mod tests { Duration::from_secs(3600), ); - let (tx, rx) = tokio::sync::oneshot::channel(); + let (tx, rx) = oneshot::channel(); started .send_checkpoint(None, CheckpointOptions::default(), tx) .unwrap(); @@ -1828,7 +1819,7 @@ mod tests { // manifest writer routes by `prefix` from the surrounding // `SegmentedSstHandle`, not by the SST's keys. let mut builder = inner.table_store.table_builder(); - let row = crate::types::RowEntry::new_value(prefix, value, first_seq); + let row = RowEntry::new_value(prefix, value, first_seq); builder.add(row).await.unwrap(); let encoded_sst = builder.build().await.unwrap(); let id = crate::db_state::SsTableId::Compacted( diff --git a/slatedb/src/memtable_flusher/mod.rs b/slatedb/src/memtable_flusher/mod.rs index c8602db624..c2221afb01 100644 --- a/slatedb/src/memtable_flusher/mod.rs +++ b/slatedb/src/memtable_flusher/mod.rs @@ -47,8 +47,8 @@ pub(crate) enum FlushTarget { /// Parallel L0 memtable flusher subsystem. pub(crate) struct MemtableFlusher { - messages_tx: SafeSender, - messages_rx: async_channel::Receiver, + messages_tx: SafeSender, + messages_rx: async_channel::Receiver, } impl MemtableFlusher { @@ -112,15 +112,14 @@ impl MemtableFlusher { pub(crate) async fn flush(&self, target: FlushTarget) -> Result { let (tx, rx) = oneshot::channel(); self.messages_tx - .send(tracker::TrackerMessage::FlushRequest { target, sender: tx })?; + .send(TrackerMessage::FlushRequest { target, sender: tx })?; rx.await.map_err(SlateDBError::ReadChannelError)? } /// Notifies the flusher that a memtable may have been frozen. /// Triggers reconcile and dispatch without waiting for a result. pub(crate) fn notify_memtable_frozen(&self) -> Result<(), SlateDBError> { - self.messages_tx - .send(tracker::TrackerMessage::MemtableFrozen) + self.messages_tx.send(TrackerMessage::MemtableFrozen) } /// Creates a checkpoint using the memtable flusher's flush semantics. @@ -130,12 +129,11 @@ impl MemtableFlusher { options: CheckpointOptions, ) -> Result { let (tx, rx) = oneshot::channel(); - self.messages_tx - .send(tracker::TrackerMessage::CheckpointRequest { - target, - options, - sender: tx, - })?; + self.messages_tx.send(TrackerMessage::CheckpointRequest { + target, + options, + sender: tx, + })?; rx.await.map_err(SlateDBError::ReadChannelError)? } diff --git a/slatedb/src/memtable_flusher/tracker.rs b/slatedb/src/memtable_flusher/tracker.rs index f608b22386..ebbff49056 100644 --- a/slatedb/src/memtable_flusher/tracker.rs +++ b/slatedb/src/memtable_flusher/tracker.rs @@ -271,7 +271,7 @@ impl FlushTracker { /// When the imm's touched-segment set is empty (no extractor /// configured, or the imm came from a path that bypassed /// validation) we fall back to the max-across-trees heuristic. - fn can_dispatch(&self, imm: &crate::mem_table::ImmutableMemtable) -> bool { + fn can_dispatch(&self, imm: &ImmutableMemtable) -> bool { let state = self.inner.state.read().state(); let core = state.core(); let settings = &self.inner.settings; @@ -413,7 +413,7 @@ fn allocate_segment_sst_ids(inner: &DbInner, imm: &ImmutableMemtable) -> BTreeMa struct TrackedImm { first_seq: u64, last_seq: u64, - imm_memtable: Arc, + imm_memtable: Arc, state: TrackedImmState, } @@ -432,10 +432,7 @@ impl TrackedImmFrontier { } /// Register newly frozen immutable memtables, deduplicating by `last_seq`. - fn register( - &mut self, - imm_memtables: impl Iterator>, - ) { + fn register(&mut self, imm_memtables: impl Iterator>) { for imm_memtable in imm_memtables { let first_seq = imm_memtable .table() diff --git a/slatedb/src/memtable_flusher/uploader.rs b/slatedb/src/memtable_flusher/uploader.rs index 271466dcd0..9865975193 100644 --- a/slatedb/src/memtable_flusher/uploader.rs +++ b/slatedb/src/memtable_flusher/uploader.rs @@ -426,12 +426,7 @@ mod tests { ) } - fn freeze_imm( - db: &DbInner, - key: &[u8], - value: &[u8], - seq: u64, - ) -> Arc { + fn freeze_imm(db: &DbInner, key: &[u8], value: &[u8], seq: u64) -> Arc { let mut guard = db.state.write(); guard.memtable().put(RowEntry::new_value(key, value, seq)); guard.freeze_memtable(0); @@ -653,11 +648,9 @@ mod tests { .await; { let mut guard = db.state.write(); - guard.memtable().put(crate::types::RowEntry::new_merge( - b"key", - b"merge_operand", - 1, - )); + guard + .memtable() + .put(RowEntry::new_merge(b"key", b"merge_operand", 1)); guard.freeze_memtable(0); } let imm_memtable = db @@ -731,7 +724,7 @@ mod tests { let mut guard = db.state.write(); guard .memtable() - .put(crate::types::RowEntry::new_merge(b"key", b"operand", 1)); + .put(RowEntry::new_merge(b"key", b"operand", 1)); guard.freeze_memtable(0); } let imm_memtable = db diff --git a/slatedb/src/merge_operator.rs b/slatedb/src/merge_operator.rs index 79c2ae5d08..7f9b27ceba 100644 --- a/slatedb/src/merge_operator.rs +++ b/slatedb/src/merge_operator.rs @@ -474,6 +474,8 @@ impl MergeOperatorIterator { } next = self.delegate.next().await?; + // Keep cached operand scans cooperative. + tokio::task::coop::consume_budget().await; } else { break None; } @@ -599,12 +601,12 @@ mod tests { /// Mock merge operator that tracks whether merge_batch is called struct MockBatchedMergeOperator { - merge_batch_call_count: std::sync::Arc, + merge_batch_call_count: Arc, } impl MockBatchedMergeOperator { - fn new() -> (Self, std::sync::Arc) { - let counter = std::sync::Arc::new(std::sync::atomic::AtomicUsize::new(0)); + fn new() -> (Self, Arc) { + let counter = Arc::new(std::sync::atomic::AtomicUsize::new(0)); ( Self { merge_batch_call_count: counter.clone(), diff --git a/slatedb/src/ops.rs b/slatedb/src/ops.rs index 8a3f80bcc6..4849275e11 100644 --- a/slatedb/src/ops.rs +++ b/slatedb/src/ops.rs @@ -242,6 +242,11 @@ pub trait DbReadOps { /// This trait defines the asynchronous write API exposed by [`Db`](crate::Db), /// allowing consumers to write generic code or test doubles over the writer /// surface without depending on the concrete `Db` type. +/// +/// Durability behavior is controlled by [`WriteOptions::await_durable`], which +/// defaults to `true`. When it is `false`, call [`WriteHandle::await_durable`] +/// on the returned handle to wait for one write, or [`Self::flush`] to flush +/// all pending writes. #[async_trait::async_trait] pub trait DbWriteOps { /// The transaction type returned by [`Self::begin`]. Stub @@ -252,6 +257,9 @@ pub trait DbWriteOps { /// Write a value into the database with default `PutOptions` and /// `WriteOptions`. /// + /// The default write options wait for durability. See [`DbWriteOps`] for + /// details. + /// /// ## Arguments /// - `key`: the key to write /// - `value`: the value to write @@ -270,6 +278,8 @@ pub trait DbWriteOps { /// Write a value into the database with custom `PutOptions` and /// `WriteOptions`. /// + /// Durability behavior follows `write_opts`. See [`DbWriteOps`] for details. + /// /// ## Arguments /// - `key`: the key to write /// - `value`: the value to write @@ -291,6 +301,9 @@ pub trait DbWriteOps { /// Delete a key from the database with default `WriteOptions`. /// + /// The default write options wait for durability. See [`DbWriteOps`] for + /// details. + /// /// ## Arguments /// - `key`: the key to delete /// @@ -303,6 +316,8 @@ pub trait DbWriteOps { /// Delete a key from the database with custom `WriteOptions`. /// + /// Durability behavior follows `options`. See [`DbWriteOps`] for details. + /// /// ## Arguments /// - `key`: the key to delete /// - `options`: the write options to use @@ -318,6 +333,9 @@ pub trait DbWriteOps { /// Merge a value into the database with default `MergeOptions` and /// `WriteOptions`. /// + /// The default write options wait for durability. See [`DbWriteOps`] for + /// details. + /// /// Merge operations allow applications to bypass the traditional /// read/modify/write cycle by expressing partial updates using an /// associative operator. The merge operator must be configured when @@ -347,6 +365,8 @@ pub trait DbWriteOps { /// Merge a value into the database with custom `MergeOptions` and /// `WriteOptions`. /// + /// Durability behavior follows `write_opts`. See [`DbWriteOps`] for details. + /// /// ## Arguments /// - `key`: the key to merge into /// - `value`: the merge operand to apply @@ -369,6 +389,9 @@ pub trait DbWriteOps { /// Write a batch of put/delete operations atomically to the database. /// + /// The default write options wait for durability. See [`DbWriteOps`] for + /// details. + /// /// ## Arguments /// - `batch`: the batch of operations to write /// @@ -382,6 +405,8 @@ pub trait DbWriteOps { /// Write a batch of put/delete operations atomically to the database with /// custom `WriteOptions`. /// + /// Durability behavior follows `options`. See [`DbWriteOps`] for details. + /// /// ## Arguments /// - `batch`: the batch of operations to write /// - `options`: the write options to use @@ -394,8 +419,8 @@ pub trait DbWriteOps { options: &WriteOptions, ) -> Result; - /// Flush in-memory writes to disk. This function blocks until the - /// in-memory data has been durably written to object storage. + /// Flush in-memory writes to object storage. This function blocks until + /// the in-memory data has been durably written. /// /// ## Errors /// - `Error`: if there was an error flushing the database. @@ -589,6 +614,10 @@ pub trait DbTransactionOps: DbReadOps { /// Commit the transaction with default `WriteOptions`. /// + /// A successful commit applies the write atomically but does not wait for + /// durability. Call [`WriteHandle::await_durable`] on the returned handle + /// when the result is `Some`. + /// /// ## Returns /// - `Ok(Some(WriteHandle))` if the commit is successful and there are /// writes in the batch. @@ -605,6 +634,11 @@ pub trait DbTransactionOps: DbReadOps { } /// Commit the transaction with custom `WriteOptions`. + /// + /// A successful commit applies the write atomically but does not wait for + /// durability. Call [`WriteHandle::await_durable`] on the returned handle + /// when the result is `Some`. + /// async fn commit_with_options( self, options: &WriteOptions, diff --git a/slatedb/src/partitioned_keyspace.rs b/slatedb/src/partitioned_keyspace.rs index 69e6911166..2da8bb0798 100644 --- a/slatedb/src/partitioned_keyspace.rs +++ b/slatedb/src/partitioned_keyspace.rs @@ -13,7 +13,7 @@ pub(crate) trait RangePartitionedKeySpace { } // equivalent to https://doc.rust-lang.org/std/primitive.slice.html#method.partition_point -fn partition_point bool>( +pub(crate) fn partition_point bool>( keyspace: &T, pred: P, ) -> usize { diff --git a/slatedb/src/paths.rs b/slatedb/src/paths.rs index a909a9615c..376b00d263 100644 --- a/slatedb/src/paths.rs +++ b/slatedb/src/paths.rs @@ -62,11 +62,11 @@ impl PathResolver { } pub(crate) fn wal_path(&self) -> Path { - Path::from(format!("{}/{}/", &self.root_path, WAL_PATH)) + Path::from(format!("{}/{}/", self.root_path, WAL_PATH)) } pub(crate) fn compacted_path(&self) -> Path { - Path::from(format!("{}/{}/", &self.root_path, COMPACTED_PATH)) + Path::from(format!("{}/{}/", self.root_path, COMPACTED_PATH)) } pub(crate) fn parse_table_id(&self, path: &Path) -> Result, SlateDBError> { @@ -76,13 +76,13 @@ impl PathResolver { .next() .and_then(|s| s.as_ref().split('.').next().map(|s| s.parse::())) .transpose() - .map(|r| r.map(SsTableId::Wal)) + .map(|r| r.map(Wal)) .map_err(|_| SlateDBError::InvalidDBState), Some(a) if a.as_ref() == COMPACTED_PATH => suffix_iter .next() .and_then(|s| s.as_ref().split('.').next().map(Ulid::from_string)) .transpose() - .map(|r| r.map(SsTableId::Compacted)) + .map(|r| r.map(Compacted)) .map_err(|_| SlateDBError::InvalidDBState), _ => Ok(None), } diff --git a/slatedb/src/reader.rs b/slatedb/src/reader.rs index 3e7c4f481b..dc3176ecbd 100644 --- a/slatedb/src/reader.rs +++ b/slatedb/src/reader.rs @@ -853,11 +853,10 @@ impl Reader { ) -> Result { self.db_stats.scan_requests.increment(1); let max_seq = self.prepare_max_seq(ctx.max_seq, options.durability_filter, options.dirty); - let read_ahead_blocks = self.table_store.bytes_to_blocks(options.read_ahead_bytes); let sst_iter_options = SstIteratorOptions { max_fetch_tasks: options.max_fetch_tasks, - blocks_to_fetch: read_ahead_blocks, + target_bytes_to_fetch: options.read_ahead_bytes, cache_blocks: options.cache_blocks, cache_metadata: true, eager_spawn: true, @@ -915,12 +914,11 @@ impl Reader { ) -> Result { self.db_stats.scan_requests.increment(1); let max_seq = self.prepare_max_seq(None, options.durability_filter, options.dirty); - let read_ahead_blocks = self.table_store.bytes_to_blocks(options.read_ahead_bytes); let range = BytesRange::from_prefix(prefix.as_ref()); let sst_iter_options = SstIteratorOptions { max_fetch_tasks: options.max_fetch_tasks, - blocks_to_fetch: read_ahead_blocks, + target_bytes_to_fetch: options.read_ahead_bytes, cache_blocks: options.cache_blocks, cache_metadata: true, // Recency scans are designed for early-stop. Eager spawning would @@ -1147,17 +1145,11 @@ mod tests { } let sst_handle = self.build_sst(entries).await?; - // Find or create the sorted run - let tree = Arc::make_mut(&mut self.core.tree); - if let Some(sr) = tree.compacted.iter_mut().find(|sr| sr.id == sr_id) { - sr.sst_views.push(SsTableView::identity(sst_handle)); - } else { - let new_sr = SortedRun { - id: sr_id, - sst_views: vec![SsTableView::identity(sst_handle)], - }; - tree.compacted.push(new_sr); - } + // The fixture groups all entries for a run into one SST before + // calling this helper, so each run is constructed exactly once. + Arc::make_mut(&mut self.core.tree) + .compacted + .push(SortedRun::new(sr_id, [SsTableView::identity(sst_handle)])); Ok(()) } @@ -1841,7 +1833,7 @@ mod tests { let write_batch = populate_db_state(&mut test_db_state, test_case.entries).await?; // Create Reader with test clock - let recorder = slatedb_common::metrics::MetricsRecorderHelper::noop(); + let recorder = MetricsRecorderHelper::noop(); let db_stats = DbStats::new(&recorder); let test_clock = Arc::new(MockSystemClock::new()); let mono_clock = Arc::new(MonotonicClock::new(test_clock as Arc, 0)); @@ -2271,7 +2263,7 @@ mod tests { let write_batch = populate_db_state(&mut test_db_state, test_case.entries).await?; // Create Reader with test clock - let recorder = slatedb_common::metrics::MetricsRecorderHelper::noop(); + let recorder = MetricsRecorderHelper::noop(); let db_stats = DbStats::new(&recorder); let test_clock = Arc::new(MockSystemClock::new()); let mono_clock = Arc::new(MonotonicClock::new(test_clock as Arc, 0)); @@ -2550,7 +2542,7 @@ mod tests { let mut test_db_state = TestDbState::new().await; let write_batch = populate_db_state(&mut test_db_state, entries).await?; - let recorder = slatedb_common::metrics::MetricsRecorderHelper::noop(); + let recorder = MetricsRecorderHelper::noop(); let db_stats = DbStats::new(&recorder); let reader = build_reader(&test_db_state, db_stats, false).await; @@ -2597,7 +2589,7 @@ mod tests { let mut test_db_state = TestDbState::new().await; let write_batch = populate_db_state(&mut test_db_state, entries).await?; - let recorder = slatedb_common::metrics::MetricsRecorderHelper::noop(); + let recorder = MetricsRecorderHelper::noop(); let db_stats = DbStats::new(&recorder); let reader = build_reader(&test_db_state, db_stats, false).await; @@ -2663,7 +2655,7 @@ mod tests { let mut test_db_state = TestDbState::new().await; let write_batch = populate_db_state(&mut test_db_state, entries).await?; - let recorder = slatedb_common::metrics::MetricsRecorderHelper::noop(); + let recorder = MetricsRecorderHelper::noop(); let db_stats = DbStats::new(&recorder); let reader = build_reader(&test_db_state, db_stats, true).await; @@ -2703,7 +2695,7 @@ mod tests { let mut test_db_state = TestDbState::new().await; let write_batch = populate_db_state(&mut test_db_state, entries).await?; - let recorder = slatedb_common::metrics::MetricsRecorderHelper::noop(); + let recorder = MetricsRecorderHelper::noop(); let db_stats = DbStats::new(&recorder); let reader = build_reader(&test_db_state, db_stats, true).await; diff --git a/slatedb/src/replay_task_scope.rs b/slatedb/src/replay_task_scope.rs index f80980b542..2186cc6630 100644 --- a/slatedb/src/replay_task_scope.rs +++ b/slatedb/src/replay_task_scope.rs @@ -15,6 +15,7 @@ pub(crate) struct ReplayTaskScope { tasks: TaskTracker, } +#[allow(dead_code)] impl ReplayTaskScope { pub(crate) fn new() -> Self { Self { diff --git a/slatedb/src/retention_iterator.rs b/slatedb/src/retention_iterator.rs index bdb1d0af8b..7407ba7b60 100644 --- a/slatedb/src/retention_iterator.rs +++ b/slatedb/src/retention_iterator.rs @@ -1028,7 +1028,7 @@ mod tests { use slatedb_common::clock::MockSystemClock; // Test the apply_retention_filter function directly since TestIterator doesn't support create_ts - let mut versions = std::collections::BTreeMap::new(); + let mut versions = BTreeMap::new(); for entry in test_case.input_entries.iter() { versions.insert(Reverse(entry.seq), entry.clone()); } @@ -1047,12 +1047,12 @@ mod tests { // Convert filtered versions back to expected order let mut actual_entries = Vec::new(); - for (_, entry) in filtered_versions.iter() { + for entry in filtered_versions.values() { actual_entries.push(entry.clone()); } // Sort by sequence number (descending) to match expected order - actual_entries.sort_by(|a, b| b.seq.cmp(&a.seq)); + actual_entries.sort_by_key(|entry| Reverse(entry.seq)); assert_eq!( actual_entries.len(), @@ -1238,7 +1238,7 @@ mod tests { assert_eq!(filtered.len(), 1); let only = filtered.values().next().unwrap(); assert!( - matches!(only.value, ValueDeletable::Tombstone), + matches!(only.value, Tombstone), "expired value should become a tombstone, got {:?}", only.value ); @@ -1318,7 +1318,7 @@ mod tests { // merge dropped, value kept as tombstone assert_eq!(filtered.len(), 1); let kept = filtered.values().next().unwrap(); - assert!(matches!(kept.value, ValueDeletable::Tombstone)); + assert!(matches!(kept.value, Tombstone)); } } } diff --git a/slatedb/src/retrying_object_store.rs b/slatedb/src/retrying_object_store.rs index b054a04a8d..ca553ed94c 100644 --- a/slatedb/src/retrying_object_store.rs +++ b/slatedb/src/retrying_object_store.rs @@ -266,7 +266,7 @@ impl ObjectStore for RetryingObjectStore { if options_range.is_none() { // No range requested — don't buffer the body. The buffer size - // can't be validated wihtout buffering. + // can't be validated without buffering. return Ok(result); } @@ -592,7 +592,7 @@ mod tests { #[tokio::test] async fn test_put_opts_retries_transient_until_success() { - let inner: Arc = Arc::new(InMemory::new()); + let inner: Arc = Arc::new(InMemory::new()); let flaky = Arc::new(FlakyObjectStore::new(inner, 1)); let retrying = RetryingObjectStore::new(flaky.clone(), test_rand(), test_clock(), None); @@ -663,7 +663,7 @@ mod tests { #[tokio::test] async fn test_put_opts_retry_sleep_uses_system_clock() { - let inner: Arc = Arc::new(InMemory::new()); + let inner: Arc = Arc::new(InMemory::new()); let flaky = Arc::new(FlakyObjectStore::new(inner, 1)); let clock = Arc::new(MockSystemClock::new()); let retrying = RetryingObjectStore::new(flaky.clone(), test_rand(), clock.clone(), None); @@ -707,7 +707,7 @@ mod tests { #[tokio::test] async fn test_put_opts_does_not_retry_on_already_exists() { - let inner: Arc = Arc::new(InMemory::new()); + let inner: Arc = Arc::new(InMemory::new()); let flaky = Arc::new(FlakyObjectStore::new(inner, 0)); let retrying = RetryingObjectStore::new(flaky.clone(), test_rand(), test_clock(), None); let path = Path::from("/data/obj"); @@ -743,7 +743,7 @@ mod tests { #[tokio::test] async fn test_head_retries_transient_until_success() { - let inner: Arc = Arc::new(InMemory::new()); + let inner: Arc = Arc::new(InMemory::new()); let path = Path::from("/x"); inner .put(&path, PutPayload::from_bytes(Bytes::from_static(b"data"))) @@ -760,7 +760,7 @@ mod tests { #[tokio::test] async fn test_put_opts_does_not_retry_on_precondition() { - let inner: Arc = Arc::new(InMemory::new()); + let inner: Arc = Arc::new(InMemory::new()); let failing = Arc::new(FlakyObjectStore::new(inner, 0).with_put_precondition_always()); let retrying = RetryingObjectStore::new(failing.clone(), test_rand(), test_clock(), None); let path = Path::from("/p"); @@ -783,7 +783,7 @@ mod tests { #[tokio::test] async fn test_get_opts_does_not_retry_on_not_modified() { - let inner: Arc = Arc::new(InMemory::new()); + let inner: Arc = Arc::new(InMemory::new()); let retrying = RetryingObjectStore::new(inner.clone(), test_rand(), test_clock(), None); let path = Path::from("/data/obj"); @@ -812,7 +812,7 @@ mod tests { #[tokio::test] async fn test_list_retries_transient_until_success() { - let inner: Arc = Arc::new(InMemory::new()); + let inner: Arc = Arc::new(InMemory::new()); let paths = [ Path::from("/items/a"), Path::from("/items/b"), @@ -847,7 +847,7 @@ mod tests { #[tokio::test] async fn test_list_with_offset_retries_transient_until_success() { - let inner: Arc = Arc::new(InMemory::new()); + let inner: Arc = Arc::new(InMemory::new()); let paths = [ Path::from("/items/a"), Path::from("/items/b"), @@ -885,7 +885,7 @@ mod tests { async fn test_put_opts_succeeds_on_matching_ulid() { // Simulate: put succeeds but returns AlreadyExists error (timeout after write) // The ULID in the object's metadata should match, so we return success - let inner: Arc = Arc::new(InMemory::new()); + let inner: Arc = Arc::new(InMemory::new()); let flaky = Arc::new( FlakyObjectStore::new(inner, 0).with_put_succeeds_but_returns_already_exists(), ); @@ -910,7 +910,7 @@ mod tests { #[tokio::test] async fn test_put_opts_fails_on_mismatched_ulid() { // First write a file with different ULID (simulating another client's write) - let inner: Arc = Arc::new(InMemory::new()); + let inner: Arc = Arc::new(InMemory::new()); let path = Path::from("/data/obj"); // Write directly to inner store (no ULID from RetryingObjectStore) @@ -948,7 +948,7 @@ mod tests { #[tokio::test] async fn test_get_range_retries_transient_until_success() { - let inner: Arc = Arc::new(InMemory::new()); + let inner: Arc = Arc::new(InMemory::new()); let path = Path::from("/data/obj"); inner .put( @@ -972,7 +972,7 @@ mod tests { #[tokio::test] async fn test_get_ranges_retries_transient_until_success() { - let inner: Arc = Arc::new(InMemory::new()); + let inner: Arc = Arc::new(InMemory::new()); let path = Path::from("/data/obj"); inner .put( @@ -1001,7 +1001,7 @@ mod tests { use object_store::{Attribute, Attributes, GetOptions}; use std::borrow::Cow; - let inner: Arc = Arc::new(InMemory::new()); + let inner: Arc = Arc::new(InMemory::new()); let retrying = RetryingObjectStore::new(inner.clone(), test_rand(), test_clock(), None); let path = Path::from("/data/obj"); @@ -1120,7 +1120,7 @@ mod tests { #[tokio::test] async fn test_bounded_max_retries_gives_up_instead_of_retrying_forever() { // Store fails more times (5) than the configured retry bound (2). - let inner: Arc = Arc::new(InMemory::new()); + let inner: Arc = Arc::new(InMemory::new()); let flaky = Arc::new(FlakyObjectStore::new(inner, 5)); let retrying = RetryingObjectStore::new(flaky.clone(), test_rand(), test_clock(), Some(2)); diff --git a/slatedb/src/segment_iterator.rs b/slatedb/src/segment_iterator.rs index 6ee26c8a2f..2929a1c12f 100644 --- a/slatedb/src/segment_iterator.rs +++ b/slatedb/src/segment_iterator.rs @@ -393,7 +393,7 @@ async fn build_sr_range_iters( let table_store = ctx.table_store.clone(); let opts = ctx.sst_iter_options.clone(); let stats = ctx.db_stats.clone(); - build_concurrent(overlapping.into_iter(), ctx.max_parallel, move |sr| { + build_concurrent(overlapping, ctx.max_parallel, move |sr| { let table_store = table_store.clone(); let range = range.clone(); let opts = opts.clone(); diff --git a/slatedb/src/seq_tracker.rs b/slatedb/src/seq_tracker.rs index 2c70ba46d4..b123c9729a 100644 --- a/slatedb/src/seq_tracker.rs +++ b/slatedb/src/seq_tracker.rs @@ -289,6 +289,7 @@ fn decode_sequence_tracker(buf: &[u8]) -> Result { let mut tracker = SequenceTracker::with_config(DEFAULT_CAPACITY, DEFAULT_INTERVAL_SECS); tracker.sequence_numbers = sequence_numbers; tracker.timestamps = timestamps; + tracker.last_recorded_ts = tracker.timestamps.last().copied(); Ok(tracker) } @@ -779,6 +780,23 @@ mod tests { assert_eq!(decoded.timestamps, tracker.timestamps); } + #[test] + fn deserialize_preserves_the_recording_interval() { + let mut tracker = SequenceTracker::new(); + tracker.insert(TrackedSeq { + seq: 0, + ts: DateTime::from_timestamp(1_600_000_060, 0).unwrap(), + }); + + let mut decoded = SequenceTracker::from_bytes(&tracker.to_bytes()).unwrap(); + decoded.insert(TrackedSeq { + seq: 1, + ts: DateTime::from_timestamp(1_600_000_061, 0).unwrap(), + }); + + assert_eq!(decoded.sequence_numbers, vec![0]); + } + #[rstest] #[case::empty_sequences(vec![])] #[case::single_sequence(vec![1000])] diff --git a/slatedb/src/size_tiered_compaction.rs b/slatedb/src/size_tiered_compaction.rs index 616ad5a2b6..0dd3c22a08 100644 --- a/slatedb/src/size_tiered_compaction.rs +++ b/slatedb/src/size_tiered_compaction.rs @@ -264,7 +264,7 @@ impl CompactionScheduler for SizeTieredCompactionScheduler { &self, state: &CompactorStateView, compaction: &CompactionSpec, - ) -> Result<(), crate::error::Error> { + ) -> Result<(), Error> { // Size-tiered does not propose drain specs and has no policy // opinions on them. Drain invariants belong to the compactor-level // validation. @@ -1011,10 +1011,7 @@ mod tests { fn create_sr(id: u32, sst_size: u64, num_ssts: usize) -> SortedRun { let ssts: Vec = (0..num_ssts).map(|_| create_sst_view(sst_size)).collect(); - SortedRun { - id, - sst_views: ssts, - } + SortedRun::new(id, ssts) } fn create_db_state(l0: VecDeque, srs: Vec) -> ManifestCore { diff --git a/slatedb/src/sorted_run_iterator.rs b/slatedb/src/sorted_run_iterator.rs index 8e7646393c..8976e494f6 100644 --- a/slatedb/src/sorted_run_iterator.rs +++ b/slatedb/src/sorted_run_iterator.rs @@ -300,10 +300,7 @@ mod tests { let encoded = builder.build().await.unwrap(); let id = SsTableId::Compacted(ulid::Ulid::new()); let handle = table_store.write_sst(&id, &encoded).await.unwrap(); - let sr = SortedRun { - id: 0, - sst_views: vec![SsTableView::identity(handle)], - }; + let sr = SortedRun::new(0, [SsTableView::identity(handle)]); let mut iter = SortedRunIterator::new_owned_initialized( .., @@ -363,13 +360,13 @@ mod tests { let encoded = builder.build().await.unwrap(); let id2 = SsTableId::Compacted(ulid::Ulid::new()); let handle2 = table_store.write_sst(&id2, &encoded).await.unwrap(); - let sr = SortedRun { - id: 0, - sst_views: vec![ + let sr = SortedRun::new( + 0, + [ SsTableView::identity(handle1), SsTableView::identity(handle2), ], - }; + ); let mut iter = SortedRunIterator::new_owned_initialized( .., @@ -435,9 +432,9 @@ mod tests { let encoded = builder.build().await.unwrap(); let id2 = SsTableId::Compacted(ulid::Ulid::new()); let handle2 = table_store.write_sst(&id2, &encoded).await.unwrap(); - let sr = SortedRun { - id: 0, - sst_views: vec![ + let sr = SortedRun::new( + 0, + [ SsTableView::new_projected( ulid::Ulid::new(), handle1, @@ -449,7 +446,7 @@ mod tests { Some(BytesRange::from_ref("key5".."key7")), ), ], - }; + ); // when: iterating the full range, then: only visible keys appear let mut iter = SortedRunIterator::new_borrowed_initialized( @@ -679,10 +676,7 @@ mod tests { ssts.push(SsTableView::identity(handle)); } - SortedRun { - id: 0, - sst_views: ssts, - } + SortedRun::new(0, ssts) } async fn build_sr_with_ssts( @@ -703,10 +697,7 @@ mod tests { let sst = writer.close().await.unwrap(); ssts.push(SsTableView::identity(sst)); } - SortedRun { - id: 0, - sst_views: ssts, - } + SortedRun::new(0, ssts) } mod mixed_version_tests { @@ -782,15 +773,15 @@ mod tests { ) .await; - let sorted_run = SortedRun { - id: 0, - sst_views: vec![ + let sorted_run = SortedRun::new( + 0, + [ SsTableView::identity(sst1_v1), SsTableView::identity(sst2_v2), SsTableView::identity(sst3_v1), SsTableView::identity(sst4_v2), ], - }; + ); // when: iterating over the sorted run let mut iter = SortedRunIterator::new_owned_initialized( @@ -855,15 +846,15 @@ mod tests { ) .await; - let sorted_run = SortedRun { - id: 0, - sst_views: vec![ + let sorted_run = SortedRun::new( + 0, + [ SsTableView::identity(sst1_v1), SsTableView::identity(sst2_v2), SsTableView::identity(sst3_v1), SsTableView::identity(sst4_v2), ], - }; + ); let mut iter = SortedRunIterator::new_owned_initialized( .., diff --git a/slatedb/src/sst_builder.rs b/slatedb/src/sst_builder.rs index e7574cdc08..24d31c073a 100644 --- a/slatedb/src/sst_builder.rs +++ b/slatedb/src/sst_builder.rs @@ -1393,7 +1393,7 @@ mod tests { let transformer = Arc::new(XorTransformer { key: 0xAB }); #[cfg(feature = "snappy")] - let compression = Some(crate::config::CompressionCodec::Snappy); + let compression = Some(CompressionCodec::Snappy); #[cfg(not(feature = "snappy"))] let compression = None; diff --git a/slatedb/src/sst_io.rs b/slatedb/src/sst_io.rs new file mode 100644 index 0000000000..54262b3aa5 --- /dev/null +++ b/slatedb/src/sst_io.rs @@ -0,0 +1,111 @@ +use std::ops::Range; +use std::sync::Arc; + +use bytes::Bytes; +use log::warn; +use object_store::path::Path; +use object_store::{Extensions, GetOptions, GetRange, ObjectStore}; + +use crate::blob::ReadOnlyBlob; +use crate::error::SlateDBError; +use crate::object_store_tag::ObjectStoreCallTag; + +/// Reads one SST object with validation retry while attaching the supplied +/// object-store call tag to every attempt. +macro_rules! read_obj { + ($object_store:expr, $path:expr, $tag:expr, |$obj:ident| $read:expr) => {{ + let object_store = $object_store; + let path = $path; + $crate::sst_io::read_with_validation_retry($tag, move |tag| { + let object_store = object_store.clone(); + let path = path.clone(); + async move { + let $obj = $crate::sst_io::ReadOnlyObject { + object_store, + path, + tag, + }; + $read.await.map_err(|error| error.with_path(&$obj.path)) + } + }) + }}; +} + +pub(crate) use read_obj; + +/// An object-store object exposed through the read-only interface consumed by +/// the shared SST format decoder. +pub(crate) struct ReadOnlyObject { + pub(crate) object_store: Arc, + pub(crate) path: Path, + pub(crate) tag: ObjectStoreCallTag, +} + +impl ReadOnlyObject { + fn extensions(&self) -> Extensions { + self.tag.into() + } +} + +impl ReadOnlyBlob for ReadOnlyObject { + async fn len(&self) -> Result { + let opts = GetOptions { + head: true, + extensions: self.extensions(), + ..GetOptions::default() + }; + let result = self.object_store.get_opts(&self.path, opts).await?; + Ok(result.meta.size) + } + + async fn read_range(&self, range: Range) -> Result { + let opts = GetOptions { + range: Some(GetRange::Bounded(range)), + extensions: self.extensions(), + ..GetOptions::default() + }; + let result = self.object_store.get_opts(&self.path, opts).await?; + Ok(result.bytes().await?) + } + + async fn read(&self) -> Result { + let opts = GetOptions { + extensions: self.extensions(), + ..GetOptions::default() + }; + let result = self.object_store.get_opts(&self.path, opts).await?; + Ok(result.bytes().await?) + } +} + +/// Number of additional attempts after an SST read fails validation. +pub(crate) const MAX_VALIDATION_RETRIES: usize = 1; + +/// Reissues recoverable validation failures with a retry reason on the object +/// store call tag. Caching object-store wrappers use that reason to invalidate +/// a corrupt local copy before the retry. +pub(crate) async fn read_with_validation_retry( + mut tag: ObjectStoreCallTag, + mut read: impl FnMut(ObjectStoreCallTag) -> Fut, +) -> Result +where + Fut: std::future::Future>, +{ + for _ in 0..MAX_VALIDATION_RETRIES { + let result = read(tag).await; + match result { + Err(ref err) => match err.maybe_validation_retry_reason() { + Some(reason) => { + warn!( + "retrying SST read after validation failure [reason={:?}, error={}]", + reason, err + ); + tag.retry = Some(reason); + } + None => return result, + }, + Ok(_) => return result, + } + } + read(tag).await +} diff --git a/slatedb/src/sst_iter.rs b/slatedb/src/sst_iter.rs index ca60761e86..a4a4e8cdfe 100644 --- a/slatedb/src/sst_iter.rs +++ b/slatedb/src/sst_iter.rs @@ -2,7 +2,6 @@ use async_trait::async_trait; use bytes::Bytes; use log::error; use slatedb_common::metrics::CounterFn; -use std::cmp::min; use std::collections::VecDeque; use std::ops::Bound::{Excluded, Included, Unbounded}; use std::ops::{Bound, Range, RangeBounds}; @@ -43,7 +42,7 @@ impl Drop for FetchTask { #[derive(Clone, Debug)] pub(crate) struct SstIteratorOptions { pub(crate) max_fetch_tasks: usize, - pub(crate) blocks_to_fetch: usize, + pub(crate) target_bytes_to_fetch: usize, pub(crate) cache_blocks: bool, pub(crate) cache_metadata: bool, pub(crate) eager_spawn: bool, @@ -56,7 +55,7 @@ impl Default for SstIteratorOptions { fn default() -> Self { SstIteratorOptions { max_fetch_tasks: 1, - blocks_to_fetch: 1, + target_bytes_to_fetch: 1, cache_blocks: true, cache_metadata: true, eager_spawn: false, @@ -91,7 +90,7 @@ impl SstView<'_> { fn point_key(&self) -> Option<&[u8]> { match (self.start_key(), self.end_key()) { - (Bound::Included(start), Bound::Included(end)) if start == end => Some(start), + (Included(start), Included(end)) if start == end => Some(start), _ => None, } } @@ -312,7 +311,7 @@ impl<'a> InternalSstIterator<'a> { replay_tasks: Option, ) -> Result { assert!(options.max_fetch_tasks > 0); - assert!(options.blocks_to_fetch > 0); + assert!(options.target_bytes_to_fetch > 0); let descending_buffer = match options.order { IterationOrder::Descending => Some(VecDeque::new()), @@ -355,6 +354,7 @@ impl<'a> InternalSstIterator<'a> { Self::new(view, table_store, options).map(Some) } + #[allow(dead_code)] fn new_owned_scoped>( range: T, table: SsTableView, @@ -415,24 +415,22 @@ impl<'a> InternalSstIterator<'a> { while self.fetch_tasks.len() < self.options.max_fetch_tasks && self.block_idx_range.contains(&self.next_block_idx_to_fetch) { - let blocks_to_fetch = min( - self.options.blocks_to_fetch, - self.block_idx_range.end - self.next_block_idx_to_fetch, - ); let table = self.view.table_as_ref().sst.clone(); + let mut blocks = self.table_store.block_range_for_target_bytes( + &table, + index, + self.next_block_idx_to_fetch, + self.options.target_bytes_to_fetch, + IterationOrder::Ascending, + ); + blocks.end = blocks.end.min(self.block_idx_range.end); let table_store = self.table_store.clone(); - let blocks_start = self.next_block_idx_to_fetch; - let blocks_end = self.next_block_idx_to_fetch + blocks_to_fetch; let index = index.clone(); let cache_blocks = self.options.cache_blocks; + let blocks_end = blocks.end; let task = async move { table_store - .read_blocks_using_index( - &table, - index, - blocks_start..blocks_end, - cache_blocks, - ) + .read_blocks_using_index(&table, index, blocks, cache_blocks) .await }; let task = if let Some(scope) = self.replay_tasks.as_ref() { @@ -449,24 +447,22 @@ impl<'a> InternalSstIterator<'a> { while self.fetch_tasks.len() < self.options.max_fetch_tasks && self.next_block_idx_to_fetch > self.block_idx_range.start { - let blocks_to_fetch = min( - self.options.blocks_to_fetch, - self.next_block_idx_to_fetch - self.block_idx_range.start, - ); let table = self.view.table_as_ref().sst.clone(); + let mut blocks = self.table_store.block_range_for_target_bytes( + &table, + index, + self.next_block_idx_to_fetch - 1, + self.options.target_bytes_to_fetch, + IterationOrder::Descending, + ); + blocks.start = blocks.start.max(self.block_idx_range.start); let table_store = self.table_store.clone(); - let blocks_end = self.next_block_idx_to_fetch; - let blocks_start = blocks_end - blocks_to_fetch; let index = index.clone(); let cache_blocks = self.options.cache_blocks; + let blocks_start = blocks.start; let task = async move { table_store - .read_blocks_using_index( - &table, - index, - blocks_start..blocks_end, - cache_blocks, - ) + .read_blocks_using_index(&table, index, blocks, cache_blocks) .await }; let task = if let Some(scope) = self.replay_tasks.as_ref() { @@ -960,6 +956,7 @@ impl<'a> SstIterator<'a> { Self::new_owned_initialized_with_stats(range, table, table_store, options, None).await } + #[allow(dead_code)] pub(crate) async fn new_owned_initialized_scoped>( range: T, table: SsTableView, @@ -1727,7 +1724,7 @@ mod tests { let sst_iter_options = SstIteratorOptions { max_fetch_tasks: 3, - blocks_to_fetch: 3, + target_bytes_to_fetch: 3 * 4096, cache_blocks: true, order, ..SstIteratorOptions::default() @@ -2011,7 +2008,7 @@ mod tests { table_store.clone(), SstIteratorOptions { max_fetch_tasks: 32, - blocks_to_fetch: 256, + target_bytes_to_fetch: 256 * 128, cache_blocks: true, cache_metadata: true, eager_spawn: false, @@ -2030,7 +2027,7 @@ mod tests { table_store.clone(), SstIteratorOptions { max_fetch_tasks: 1, - blocks_to_fetch: 1, + target_bytes_to_fetch: 1, cache_blocks: true, cache_metadata: true, eager_spawn: false, @@ -2617,7 +2614,7 @@ mod tests { let sst_iter_options = SstIteratorOptions { max_fetch_tasks: 3, - blocks_to_fetch: 3, + target_bytes_to_fetch: 3 * 128, cache_blocks: true, cache_metadata: true, eager_spawn: false, @@ -2910,7 +2907,7 @@ mod tests { table_store.clone(), SstIteratorOptions { max_fetch_tasks: 1, - blocks_to_fetch: 1, + target_bytes_to_fetch: 1, cache_blocks: true, cache_metadata: true, eager_spawn: false, diff --git a/slatedb/src/sst_reader.rs b/slatedb/src/sst_reader.rs index e818c1f73b..c4dd2aeabf 100644 --- a/slatedb/src/sst_reader.rs +++ b/slatedb/src/sst_reader.rs @@ -47,7 +47,6 @@ use std::sync::Arc; -use bytes::Bytes; use object_store::path::Path; use object_store::ObjectStore; use ulid::Ulid; @@ -56,9 +55,11 @@ use crate::block_cache_policy::BlockCachePolicy; use crate::block_iterator::DataBlockIterator; use crate::db_cache::DbCache; use crate::db_state::{SsTableHandle, SsTableId, SsTableInfo}; +use crate::flatbuffer_types::SsTableIndexOwned; use crate::format::sst::{BlockTransformer, SsTableFormat}; use crate::iter::IterationOrder; use crate::object_stores::ObjectStores; +use crate::partitioned_keyspace::{partition_point, RangePartitionedKeySpace}; use crate::sst_stats::SstStats; use crate::tablestore::{TableStore, TableStoreKind}; use crate::types::RowEntry; @@ -147,6 +148,69 @@ pub struct SstFile { table_store: Arc, } +/// A zero-copy view of an SST's block index. +/// +/// The view owns a reference to the cached index data. Keys returned by its +/// accessors borrow directly from that data without allocation or copying. +#[derive(Clone)] +pub struct SstIndex { + inner: Arc, +} + +impl SstIndex { + /// Returns the number of data blocks described by the index. + pub fn len(&self) -> usize { + self.inner.borrow().block_meta().len() + } + + /// Returns whether the index contains no data blocks. + pub fn is_empty(&self) -> bool { + self.len() == 0 + } + + /// Returns the block offset and first key at `index`. + pub fn get(&self, index: usize) -> Option<(u64, &[u8])> { + let block_meta = self.inner.borrow().block_meta(); + if index >= block_meta.len() { + return None; + } + let meta = block_meta.get(index); + Some((meta.offset(), meta.first_key().bytes())) + } + + /// Iterates over block offsets and first keys in index order. + pub fn iter(&self) -> impl ExactSizeIterator + '_ { + let block_meta = self.inner.borrow().block_meta(); + (0..block_meta.len()).map(move |index| { + let meta = block_meta.get(index); + (meta.offset(), meta.first_key().bytes()) + }) + } + + /// Returns the first index for which `pred` is false. + /// + /// The index first keys are sorted, so `pred` must return `true` for a + /// contiguous prefix of the index, matching [`slice::partition_point`]. + pub fn partition_point

(&self, pred: P) -> usize + where + P: Fn(&[u8]) -> bool, + { + partition_point(self, pred) + } +} + +impl RangePartitionedKeySpace for SstIndex { + fn partitions(&self) -> usize { + self.len() + } + + fn partition_first_key(&self, partition: usize) -> &[u8] { + self.get(partition) + .expect("partition index should be in range") + .1 + } +} + impl SstFile { /// Returns the SST's ULID identifier. pub fn id(&self) -> Ulid { @@ -192,29 +256,20 @@ impl SstFile { .map_err(Into::into) } - /// Returns `(block_offset, first_key)` pairs from the SST index block. + /// Returns a zero-copy view of the SST index block. /// - /// The returned vector is parallel to the data blocks in the SST. Each - /// entry contains the on-disk byte offset of the block and the first key - /// stored in that block. + /// The returned [`SstIndex`] contains one entry for each data block in the + /// SST, in block order. Each entry contains the on-disk byte offset of + /// the block and the first key stored in that block. The index keeps + /// the cached index data alive, and keys returned by its accessors borrow + /// directly from that data without allocation or copying. /// /// ## Errors /// /// Returns an error if there is an issue reading from object storage. - pub async fn index(&self) -> Result, crate::Error> { - let index = self.table_store.read_index(&self.handle, true).await?; - let borrowed = index.borrow(); - let block_meta = borrowed.block_meta(); - let result: Vec<(u64, Bytes)> = (0..block_meta.len()) - .map(|i| { - let meta = block_meta.get(i); - ( - meta.offset(), - Bytes::copy_from_slice(meta.first_key().bytes()), - ) - }) - .collect(); - Ok(result) + pub async fn index(&self) -> Result { + let inner = self.table_store.read_index(&self.handle, true).await?; + Ok(SstIndex { inner }) } /// Reads a single data block by its index and returns the decoded rows. @@ -262,6 +317,7 @@ mod tests { use crate::test_utils::StringConcatMergeOperator; use crate::types::ValueDeletable; use crate::Db; + use bytes::Bytes; use object_store::memory::InMemory; /// Helper: create a DB with 10 puts, 3 deletes, and 2 merges, flush to @@ -376,11 +432,41 @@ mod tests { // First index key should be <= the SST's first entry (it may be a // shortened separator key rather than the exact first key). if let Some(first_entry) = sst_file.info().first_entry.as_ref() { - assert!(index[0].1.as_ref() <= first_entry.as_ref()); + let (_, first_key) = index.get(0).expect("index should not be empty"); + assert!(first_key <= first_entry.as_ref()); } // Offsets should be monotonically increasing - for window in index.windows(2) { - assert!(window[0].0 < window[1].0); + let mut entries = index.iter(); + let mut previous_offset = entries.next().expect("index should not be empty").0; + for (offset, _) in entries { + assert!(previous_offset < offset); + previous_offset = offset; + } + assert_eq!(index.get(index.len()), None); + } + + #[tokio::test] + async fn test_index_partition_point_matches_slice() { + let (store, path, manifest) = setup_db_with_l0().await; + let reader = SstReader::new(path, store, None, None); + + let view = &manifest.manifest.core.tree.l0[0]; + let sst_file = reader.open_with_handle(view.sst.clone()).unwrap(); + let index = sst_file.index().await.unwrap(); + let owned = index + .iter() + .map(|(offset, first_key)| (offset, Bytes::copy_from_slice(first_key))) + .collect::>(); + + for key in [b"".as_slice(), b"k00", b"k05", b"k99"] { + assert_eq!( + index.partition_point(|candidate| candidate < key), + owned.partition_point(|(_, candidate)| candidate.as_ref() < key) + ); + assert_eq!( + index.partition_point(|candidate| candidate <= key), + owned.partition_point(|(_, candidate)| candidate.as_ref() <= key) + ); } } @@ -501,11 +587,7 @@ mod tests { let store: Arc = Arc::new(InMemory::new()); let reader = SstReader::new("/test", store, None, None); - let wal_handle = SsTableHandle::new( - SsTableId::Wal(42), - 0, - crate::db_state::SsTableInfo::default(), - ); + let wal_handle = SsTableHandle::new(SsTableId::Wal(42), 0, SsTableInfo::default()); let result = reader.open_with_handle(wal_handle); assert!(result.is_err()); } diff --git a/slatedb/src/sst_stats.rs b/slatedb/src/sst_stats.rs index 7fe3d9dce9..727bda2712 100644 --- a/slatedb/src/sst_stats.rs +++ b/slatedb/src/sst_stats.rs @@ -40,7 +40,7 @@ impl SstStats { /// Returns the in-memory size in bytes (struct + heap-allocated block_stats). pub(crate) fn size(&self) -> usize { - std::mem::size_of::() + self.block_stats.len() * std::mem::size_of::() + size_of::() + self.block_stats.len() * size_of::() } /// Returns a clone. diff --git a/slatedb/src/subcompaction.rs b/slatedb/src/subcompaction.rs index 2abaa11a2a..085f900766 100644 --- a/slatedb/src/subcompaction.rs +++ b/slatedb/src/subcompaction.rs @@ -121,7 +121,7 @@ pub(crate) async fn plan_subcompaction_ranges( .chain( sorted_runs .iter() - .flat_map(|sr| sr.sst_views.iter().cloned()), + .flat_map(|sr| sr.sst_views().iter().cloned()), ) .collect(); diff --git a/slatedb/src/tablestore.rs b/slatedb/src/tablestore.rs index 4e5b007bee..7d8f75b425 100644 --- a/slatedb/src/tablestore.rs +++ b/slatedb/src/tablestore.rs @@ -8,35 +8,33 @@ use futures::{future::join_all, StreamExt}; use log::{debug, warn}; use object_store::buffered::BufWriter; use object_store::path::Path; -use object_store::{ - Extensions, GetOptions, GetRange, ObjectStore, ObjectStoreExt, PutMode, PutOptions, -}; +use object_store::{GetOptions, ObjectStore, ObjectStoreExt, PutMode, PutOptions}; use slatedb_common::object_metadata::IdentifiedObjectMetadata; use slatedb_common::ObjectMetadata; use tokio::io::AsyncWriteExt; use ulid::Ulid; -use crate::blob::{BytesBlob, ReadOnlyBlob}; use crate::block_cache_policy::{should_cache_data_block, BlockCachePolicy}; use crate::db_cache::CacheTarget; use crate::db_cache::{CacheLoader, CachedEntry, CachedKey, DbCache, EncodedCachedFilter}; -use crate::db_state::{SsTableHandle, SsTableId, SsTableInfo, SstType}; +use crate::db_state::{SsTableHandle, SsTableId, SstType}; use crate::error::SlateDBError; use crate::filter_policy::NamedFilter; use crate::flatbuffer_types::SsTableIndexOwned; use crate::format::block::Block; -use crate::format::sst::{ - EncodedSsTable, EncodedSsTableBlock, SsTableFormat, StagedSstInfoError, CHECKSUM_SIZE, - METADATA_OFFSET_SIZE, VERSION_SIZE, -}; +use crate::format::sst::{EncodedSsTable, EncodedSsTableBlock, SsTableFormat}; +use crate::iter::IterationOrder; use crate::object_store_tag::ObjectStoreCallTag; pub(crate) use crate::object_store_tag::TableStoreKind; use crate::object_stores::{ObjectStoreType, ObjectStores}; use crate::paths::PathResolver; use crate::sst_builder::EncodedSsTableBuilder; +#[cfg(test)] +use crate::sst_io::MAX_VALIDATION_RETRIES; +use crate::sst_io::{read_obj as read_sst_obj, read_with_validation_retry, ReadOnlyObject}; use crate::sst_stats::SstStats; use crate::types::RowEntry; -use crate::wal::wal_sst_builder::EncodedWalSsTableBuilder; +use crate::wal::slatedb::sst_builder::EncodedWalSsTableBuilder; pub(crate) struct TableStore { object_stores: ObjectStores, @@ -52,110 +50,6 @@ pub(crate) struct TableStore { kind: TableStoreKind, } -pub(crate) enum DecodedWalSst { - Fence, - Data(Box), -} - -pub(crate) struct DecodedWalSstData { - pub(crate) wal_id: u64, - pub(crate) format_version: u16, - pub(crate) object_bytes: Bytes, - pub(crate) info: SsTableInfo, - pub(crate) index: SsTableIndexOwned, - pub(crate) retained_decode_bytes: usize, -} - -pub(crate) enum RangedWalSst { - Fence, - Data(Box), -} - -pub(crate) struct RangedWalSstData { - pub(crate) wal_id: u64, - pub(crate) format_version: u16, - pub(crate) info: SsTableInfo, - pub(crate) index: SsTableIndexOwned, - pub(crate) retained_decode_bytes: usize, -} - -pub(crate) enum RuntimeWalOpenError { - MissingInitialObject(SlateDBError), - Replay(SlateDBError), -} - -struct ReadOnlyObject { - object_store: Arc, - path: Path, - tag: ObjectStoreCallTag, -} - -impl ReadOnlyObject { - fn extensions(&self) -> Extensions { - self.tag.into() - } -} - -/// Reads from a [`ReadOnlyObject`] for an SST `$id`, with validation-retry. -/// -/// It expands to the retry-wrapper future, so callers `.await` it. -/// This is used instead of repeating the same retry logic for every individual -/// read from an SST object. -macro_rules! read_obj { - ($store:expr, $id:expr, |$obj:ident| $read:expr) => {{ - let object_store = $store.object_stores.store_for($id); - let path = $store.path($id); - read_with_validation_retry( - ObjectStoreCallTag::new($store.kind, SstType::from($id)), - move |tag| { - let object_store = object_store.clone(); - let path = path.clone(); - async move { - let $obj = ReadOnlyObject { - object_store, - path, - tag, - }; - $read.await.map_err(|e| e.with_path(&$obj.path)) - } - }, - ) - }}; -} - -impl ReadOnlyBlob for ReadOnlyObject { - async fn len(&self) -> Result { - let opts = GetOptions { - head: true, - extensions: self.extensions(), - ..GetOptions::default() - }; - let result = self.object_store.get_opts(&self.path, opts).await?; - Ok(result.meta.size) - } - - async fn read_range(&self, range: Range) -> Result { - let opts = GetOptions { - range: Some(GetRange::Bounded(range)), - extensions: self.extensions(), - ..GetOptions::default() - }; - let result = self.object_store.get_opts(&self.path, opts).await?; - let bytes = result.bytes().await?; - Ok(bytes) - } - - async fn read(&self) -> Result { - let opts = GetOptions { - extensions: self.extensions(), - ..GetOptions::default() - }; - let result = self.object_store.get_opts(&self.path, opts).await?; - let bytes = result.bytes().await?; - Ok(bytes) - } -} - impl TableStore { pub(crate) fn new>( object_stores: ObjectStores, @@ -196,135 +90,6 @@ impl TableStore { } } - /// Get the number of blocks for a size specified in bytes. - /// The returned value will be rounded down to the nearest block. - pub(crate) fn bytes_to_blocks(&self, bytes: usize) -> usize { - bytes.div_ceil(self.sst_format.block_size) - } - - pub(crate) fn validate_wal_sst_replay_memory( - &self, - encoded_sst: &EncodedSsTable, - metadata_memory_limit: usize, - block_memory_limit: usize, - ) -> Result<(), SlateDBError> { - let index_encoded_bytes = usize::try_from(encoded_sst.info.index_len).map_err(|_| { - SlateDBError::WalReplayMemoryLimitExceeded { - kind: "encoded and decoded WAL index", - required_bytes: usize::MAX, - limit_bytes: metadata_memory_limit, - } - })?; - let metadata_encoded_bytes = encoded_sst - .footer - .len() - .checked_sub(METADATA_OFFSET_SIZE + VERSION_SIZE) - .and_then(|footer_bytes| footer_bytes.checked_sub(index_encoded_bytes)) - .ok_or(SlateDBError::InvalidDBState)?; - let retained_info_bytes = encoded_sst - .info - .first_entry - .as_ref() - .map_or(0, Bytes::len) - .checked_add(encoded_sst.info.last_entry.as_ref().map_or(0, Bytes::len)) - .ok_or(SlateDBError::WalReplayMemoryLimitExceeded { - kind: "encoded and decoded WAL metadata", - required_bytes: usize::MAX, - limit_bytes: metadata_memory_limit, - })?; - let metadata_required_bytes = metadata_encoded_bytes - .checked_add(retained_info_bytes) - .ok_or(SlateDBError::WalReplayMemoryLimitExceeded { - kind: "encoded and decoded WAL metadata", - required_bytes: usize::MAX, - limit_bytes: metadata_memory_limit, - })?; - if metadata_required_bytes > metadata_memory_limit { - return Err(SlateDBError::WalReplayMemoryLimitExceeded { - kind: "encoded and decoded WAL metadata", - required_bytes: metadata_required_bytes, - limit_bytes: metadata_memory_limit, - }); - } - - let index_payload_bytes = index_encoded_bytes - .checked_sub(CHECKSUM_SIZE) - .ok_or(SlateDBError::InvalidDBState)?; - let transformed_index_bytes = match &self.sst_format.block_transformer { - Some(transformer) => transformer.max_decoded_len(index_payload_bytes).ok_or( - SlateDBError::WalReplayMemoryLimitExceeded { - kind: "transformed WAL index", - required_bytes: usize::MAX, - limit_bytes: metadata_memory_limit, - }, - )?, - None => 0, - }; - let index_required_bytes = retained_info_bytes - .checked_add(index_encoded_bytes) - .and_then(|bytes| bytes.checked_add(transformed_index_bytes)) - .and_then(|bytes| bytes.checked_add(encoded_sst.index.size())) - .ok_or(SlateDBError::WalReplayMemoryLimitExceeded { - kind: "encoded and decoded WAL index", - required_bytes: usize::MAX, - limit_bytes: metadata_memory_limit, - })?; - if index_required_bytes > metadata_memory_limit { - return Err(SlateDBError::WalReplayMemoryLimitExceeded { - kind: "encoded and decoded WAL index", - required_bytes: index_required_bytes, - limit_bytes: metadata_memory_limit, - }); - } - - for block in &encoded_sst.unconsumed_blocks { - let encoded_payload_bytes = block - .encoded_bytes - .len() - .checked_sub(CHECKSUM_SIZE) - .ok_or(SlateDBError::InvalidDBState)?; - let transformed_block_bytes = match &self.sst_format.block_transformer { - Some(transformer) => transformer.max_decoded_len(encoded_payload_bytes).ok_or( - SlateDBError::WalReplayMemoryLimitExceeded { - kind: "transformed WAL block", - required_bytes: usize::MAX, - limit_bytes: block_memory_limit, - }, - )?, - None => 0, - }; - let offsets_bytes = block - .block - .offsets - .len() - .checked_mul(std::mem::size_of::()) - .ok_or(SlateDBError::WalReplayMemoryLimitExceeded { - kind: "encoded and decoded WAL block", - required_bytes: usize::MAX, - limit_bytes: block_memory_limit, - })?; - let required_bytes = block - .encoded_bytes - .len() - .checked_add(transformed_block_bytes) - .and_then(|bytes| bytes.checked_add(block.block.size())) - .and_then(|bytes| bytes.checked_add(offsets_bytes)) - .ok_or(SlateDBError::WalReplayMemoryLimitExceeded { - kind: "encoded and decoded WAL block", - required_bytes: usize::MAX, - limit_bytes: block_memory_limit, - })?; - if required_bytes > block_memory_limit { - return Err(SlateDBError::WalReplayMemoryLimitExceeded { - kind: "encoded and decoded WAL block", - required_bytes, - limit_bytes: block_memory_limit, - }); - } - } - Ok(()) - } - /// Find the highest WAL SST id present in the object store at or above /// `start_after + 1`, returning `start_after` if none exist. /// @@ -342,6 +107,7 @@ impl TableStore { /// Relies on the fencing protocol's contiguity invariant: "id exists" is /// monotone-decreasing in id, so binary search is sound. Total HEAD count /// is `O(log N)` for a gap of size N, vs `O(N)` for a windowed scan. + #[allow(unused)] pub(crate) async fn last_seen_wal_id(&self, start_after: u64) -> Result { fail_point!(Arc::clone(&self.fp_registry), "probe-wal-ssts", |_| { Err(SlateDBError::from(std::io::Error::other("oops"))) @@ -478,421 +244,7 @@ impl TableStore { Ok(wal_list) } - /// Lists only the WAL suffix needed for one replay operation. - pub(crate) async fn list_wal_ssts_for_replay( - &self, - id_range: Range, - ) -> Result>, SlateDBError> { - if id_range.is_empty() { - return Ok(Vec::new()); - } - - let wal_path = self.path_resolver.wal_path(); - let object_store = self.object_stores.store_of(ObjectStoreType::Wal); - let mut files_stream = if id_range.start == 0 { - object_store.list(Some(&wal_path)) - } else { - let offset = self.path(&SsTableId::Wal(id_range.start - 1)); - object_store.list_with_offset(Some(&wal_path), &offset) - }; - let mut wal_list = Vec::new(); - - while let Some(file) = files_stream.next().await.transpose()? { - let Ok(Some(SsTableId::Wal(id))) = self.path_resolver.parse_table_id(&file.location) - else { - continue; - }; - if id >= id_range.end { - break; - } - if id >= id_range.start { - wal_list.push(IdentifiedObjectMetadata::from_object_meta( - SsTableId::Wal(id), - file, - )); - } - } - wal_list.sort_by_key(|metadata| metadata.id.unwrap_wal_id()); - Ok(wal_list) - } - - /// Fetches a WAL SST with one full-object GET. - pub(crate) async fn read_wal_sst_bytes( - &self, - wal_id: u64, - expected_size: Option, - max_object_bytes: usize, - ) -> Result { - self.read_wal_sst_bytes_with_retry(wal_id, expected_size, max_object_bytes, None) - .await - } - - async fn read_wal_sst_bytes_with_retry( - &self, - wal_id: u64, - expected_size: Option, - max_object_bytes: usize, - retry: Option, - ) -> Result { - let id = SsTableId::Wal(wal_id); - let obj = ReadOnlyObject { - object_store: self.object_stores.store_for(&id), - path: self.path(&id), - tag: ObjectStoreCallTag::new(self.kind, SstType::Wal), - }; - let mut tag = ObjectStoreCallTag::new(self.kind, SstType::Wal); - tag.retry = retry; - // Some object stores reject a body read for an existing zero-byte object. - // A tagged HEAD still proves that the exact fence object exists and has - // the size observed by the replay LIST. - let is_expected_fence = expected_size == Some(0); - let opts = GetOptions { - head: is_expected_fence, - extensions: tag.into(), - ..GetOptions::default() - }; - let result = obj - .object_store - .get_opts(&obj.path, opts) - .await - .map_err(SlateDBError::from) - .map_err(|err| err.with_path(&obj.path))?; - let response_size = usize::try_from(result.meta.size).map_err(|_| { - SlateDBError::WalReplayMemoryLimitExceeded { - kind: "encoded WAL object", - required_bytes: usize::MAX, - limit_bytes: max_object_bytes, - } - })?; - if response_size > max_object_bytes { - return Err(SlateDBError::WalReplayMemoryLimitExceeded { - kind: "encoded WAL object", - required_bytes: response_size, - limit_bytes: max_object_bytes, - }); - } - if expected_size.is_some_and(|expected| result.meta.size != expected) { - return Err(invalid_wal_size( - &obj.path, - wal_id, - expected_size, - result.meta.size, - )); - } - if is_expected_fence { - return Ok(Bytes::new()); - } - let response_size_u64 = result.meta.size; - let bytes = result - .bytes() - .await - .map_err(SlateDBError::from) - .map_err(|err| err.with_path(&obj.path))?; - let actual_size = u64::try_from(bytes.len()).map_err(|err| { - SlateDBError::WalDataError(Arc::new(std::io::Error::new( - std::io::ErrorKind::InvalidData, - err, - ))) - })?; - if actual_size != response_size_u64 { - return Err(invalid_wal_size( - &obj.path, - wal_id, - Some(response_size_u64), - actual_size, - )); - } - Ok(bytes) - } - - pub(crate) async fn refetch_wal_sst_after_validation( - &self, - wal_id: u64, - expected_size: usize, - max_object_bytes: usize, - working_memory_limit: usize, - reason: crate::error::RetryReason, - ) -> Result { - let expected_size = u64::try_from(expected_size).map_err(|_| { - SlateDBError::WalReplayMemoryLimitExceeded { - kind: "encoded WAL object", - required_bytes: expected_size, - limit_bytes: max_object_bytes, - } - })?; - let bytes = self - .read_wal_sst_bytes_with_retry( - wal_id, - Some(expected_size), - max_object_bytes, - Some(reason), - ) - .await?; - self.decode_wal_sst(wal_id, bytes, working_memory_limit) - .await - } - - /// Decodes the metadata and index of a fully fetched WAL locally. Data - /// blocks are decoded lazily by [`Self::decode_wal_block`]. - pub(crate) async fn decode_wal_sst( - &self, - wal_id: u64, - bytes: Bytes, - working_memory_limit: usize, - ) -> Result { - let metadata_memory_limit = working_memory_limit / 2; - let path = self.path(&SsTableId::Wal(wal_id)); - if bytes.is_empty() { - return Ok(DecodedWalSst::Fence); - } - let blob = BytesBlob::new(bytes.clone()); - let object_len = blob.len().await?; - let footer_size = METADATA_OFFSET_SIZE + VERSION_SIZE; - let footer_start = bytes - .len() - .checked_sub(footer_size) - .ok_or(SlateDBError::EmptySSTable)?; - let metadata_offset = u64::from_be_bytes( - bytes[footer_start..footer_start + METADATA_OFFSET_SIZE] - .try_into() - .map_err(|_| SlateDBError::EmptySSTable)?, - ); - let (info, format_version) = self - .sst_format - .read_info_and_version(&blob) - .await - .map_err(|error| error.with_path(&path))?; - validate_wal_sst_info_layout(&info, object_len, metadata_offset, &path)?; - let retained_info_bytes = info - .first_entry - .as_ref() - .map_or(0, Bytes::len) - .checked_add(info.last_entry.as_ref().map_or(0, Bytes::len)) - .ok_or(SlateDBError::WalReplayMemoryLimitExceeded { - kind: "WAL metadata and index", - required_bytes: usize::MAX, - limit_bytes: metadata_memory_limit, - })?; - let index_memory_limit = metadata_memory_limit - .checked_sub(retained_info_bytes) - .ok_or(SlateDBError::WalReplayMemoryLimitExceeded { - kind: "WAL metadata and index", - required_bytes: retained_info_bytes, - limit_bytes: metadata_memory_limit, - })?; - let index = self - .sst_format - .read_wal_index_bounded(&info, &blob, index_memory_limit) - .await - .map_err(|error| error.with_path(&path))?; - validate_wal_sst_index_layout(&info, &index, &path)?; - let retained_decode_bytes = index - .size() - .checked_add(info.first_entry.as_ref().map_or(0, Bytes::len)) - .and_then(|size| size.checked_add(info.last_entry.as_ref().map_or(0, Bytes::len))) - .ok_or(SlateDBError::WalReplayMemoryLimitExceeded { - kind: "WAL metadata and index", - required_bytes: usize::MAX, - limit_bytes: metadata_memory_limit, - })?; - if retained_decode_bytes > metadata_memory_limit { - return Err(SlateDBError::WalReplayMemoryLimitExceeded { - kind: "WAL metadata and index", - required_bytes: retained_decode_bytes, - limit_bytes: metadata_memory_limit, - }); - } - Ok(DecodedWalSst::Data(Box::new(DecodedWalSstData { - wal_id, - format_version, - object_bytes: bytes, - info, - index, - retained_decode_bytes, - }))) - } - - /// Opens an oversized WAL without retaining its encoded object. Metadata, - /// the index, and each data block are fetched with bounded range reads. - pub(crate) async fn open_ranged_wal_sst( - &self, - wal_id: u64, - expected_size: u64, - working_memory_limit: usize, - ) -> Result { - let metadata_memory_limit = working_memory_limit / 2; - let id = SsTableId::Wal(wal_id); - read_obj!(self, &id, |obj| async { - let object_len = obj.len().await?; - if object_len != expected_size { - return Err(invalid_wal_size( - &obj.path, - wal_id, - Some(expected_size), - object_len, - )); - } - if object_len == 0 { - return Ok(RangedWalSst::Fence); - } - - let footer_size = u64::try_from(METADATA_OFFSET_SIZE + VERSION_SIZE) - .map_err(|_| SlateDBError::InvalidDBState)?; - let footer_start = object_len - .checked_sub(footer_size) - .ok_or(SlateDBError::EmptySSTable)?; - let footer = obj.read_range(footer_start..object_len).await?; - let metadata_offset = u64::from_be_bytes( - footer - .get(..METADATA_OFFSET_SIZE) - .ok_or(SlateDBError::EmptySSTable)? - .try_into() - .map_err(|_| SlateDBError::EmptySSTable)?, - ); - let version = u16::from_be_bytes( - footer - .get(METADATA_OFFSET_SIZE..METADATA_OFFSET_SIZE + VERSION_SIZE) - .ok_or(SlateDBError::EmptySSTable)? - .try_into() - .map_err(|_| SlateDBError::EmptySSTable)?, - ); - let (info, format_version) = self - .sst_format - .read_wal_info_and_version_bounded( - &obj, - object_len, - metadata_offset, - version, - metadata_memory_limit, - ) - .await?; - validate_wal_sst_info_layout(&info, object_len, metadata_offset, &obj.path)?; - let retained_info_bytes = info - .first_entry - .as_ref() - .map_or(0, Bytes::len) - .checked_add(info.last_entry.as_ref().map_or(0, Bytes::len)) - .ok_or(SlateDBError::WalReplayMemoryLimitExceeded { - kind: "WAL metadata and index", - required_bytes: usize::MAX, - limit_bytes: metadata_memory_limit, - })?; - let index_memory_limit = metadata_memory_limit - .checked_sub(retained_info_bytes) - .ok_or(SlateDBError::WalReplayMemoryLimitExceeded { - kind: "WAL metadata and index", - required_bytes: retained_info_bytes, - limit_bytes: metadata_memory_limit, - })?; - let index = self - .sst_format - .read_ranged_wal_index_bounded(&info, &obj, index_memory_limit) - .await?; - validate_wal_sst_index_layout(&info, &index, &obj.path)?; - let retained_decode_bytes = index - .size() - .checked_add(info.first_entry.as_ref().map_or(0, Bytes::len)) - .and_then(|size| size.checked_add(info.last_entry.as_ref().map_or(0, Bytes::len))) - .ok_or(SlateDBError::WalReplayMemoryLimitExceeded { - kind: "WAL metadata and index", - required_bytes: usize::MAX, - limit_bytes: metadata_memory_limit, - })?; - if retained_decode_bytes > metadata_memory_limit { - return Err(SlateDBError::WalReplayMemoryLimitExceeded { - kind: "WAL metadata and index", - required_bytes: retained_decode_bytes, - limit_bytes: metadata_memory_limit, - }); - } - Ok(RangedWalSst::Data(Box::new(RangedWalSstData { - wal_id, - format_version, - info, - index, - retained_decode_bytes, - }))) - }) - .await - } - - pub(crate) async fn decode_wal_block( - &self, - wal: &DecodedWalSstData, - block_index: usize, - working_memory_limit: usize, - ) -> Result { - let path = self.path(&SsTableId::Wal(wal.wal_id)); - self.sst_format - .decode_wal_block_from_object( - &wal.info, - &wal.index, - block_index, - &wal.object_bytes, - working_memory_limit, - ) - .await - .map_err(|error| error.with_path(&path)) - } - - pub(crate) async fn read_ranged_wal_block( - &self, - wal: &RangedWalSstData, - block_index: usize, - working_memory_limit: usize, - ) -> Result { - let (start, end) = { - let index = wal.index.borrow(); - if block_index >= index.block_meta().len() { - return Err(SlateDBError::CorruptSst { - reason: "WAL block index is out of range", - path: None, - }); - } - let start = index.block_meta().get(block_index).offset(); - let end = if block_index + 1 < index.block_meta().len() { - index.block_meta().get(block_index + 1).offset() - } else { - wal.info.filter_offset - }; - (start, end) - }; - if start >= end { - return Err(SlateDBError::CorruptSst { - reason: "WAL block range is empty or reversed", - path: None, - }); - } - let expected_len = usize::try_from(end - start).map_err(|_| { - SlateDBError::WalReplayMemoryLimitExceeded { - kind: "encoded WAL block", - required_bytes: usize::MAX, - limit_bytes: working_memory_limit, - } - })?; - let decode_memory_limit = working_memory_limit.checked_sub(expected_len).ok_or( - SlateDBError::WalReplayMemoryLimitExceeded { - kind: "encoded and decoded WAL block", - required_bytes: expected_len, - limit_bytes: working_memory_limit, - }, - )?; - let id = SsTableId::Wal(wal.wal_id); - read_obj!(self, &id, |obj| async { - let bytes = obj.read_range(start..end).await?; - if bytes.len() != expected_len { - return Err(SlateDBError::CorruptSst { - reason: "WAL block range returned an unexpected length", - path: None, - }); - } - self.sst_format - .decode_wal_block_bounded(bytes, wal.info.compression_codec, decode_memory_limit) - .await - }) - .await - } - + #[allow(unused)] pub(crate) async fn next_wal_sst_id( &self, wal_id_last_compacted: u64, @@ -921,6 +273,7 @@ impl TableStore { self.sst_format.table_builder() } + #[allow(unused)] pub(crate) fn wal_table_builder(&self) -> EncodedWalSsTableBuilder { self.sst_format.wal_table_builder() } @@ -934,13 +287,13 @@ impl TableStore { self.fp_registry.clone(), "write-wal-sst-io-error", matches!(id, SsTableId::Wal(_)), - |_| Result::Err(slatedb_io_error()) + |_| Err(slatedb_io_error()) ); fail_point!( self.fp_registry.clone(), "write-compacted-sst-io-error", matches!(id, SsTableId::Compacted(_)), - |_| Result::Err(slatedb_io_error()) + |_| Err(slatedb_io_error()) ); let object_store = self.object_stores.store_for(id); @@ -1053,10 +406,11 @@ impl TableStore { /// /// Uses create-if-absent semantics so any existing WAL object at this ID /// fences the writer by returning [`SlateDBError::Fenced`]. + #[allow(unused)] pub(crate) async fn write_wal_fence(&self, wal_id: u64) -> Result<(), SlateDBError> { let id = SsTableId::Wal(wal_id); fail_point!(self.fp_registry.clone(), "write-wal-sst-io-error", |_| { - Result::Err(slatedb_io_error()) + Err(slatedb_io_error()) }); write_sst_in_object_store( self.object_stores.store_for(&id), @@ -1186,61 +540,25 @@ impl TableStore { } pub(crate) async fn open_sst(&self, id: &SsTableId) -> Result { - let (info, version) = - read_obj!(self, id, |obj| self.sst_format.read_info_and_version(&obj)).await?; + let (info, version) = read_sst_obj!( + self.object_stores.store_for(id), + self.path(id), + ObjectStoreCallTag::new(self.kind, SstType::from(id)), + |obj| self.sst_format.read_info_and_version(&obj) + ) + .await?; Ok(SsTableHandle::new(*id, version, info)) } - pub(crate) async fn open_runtime_wal_sst( - &self, - wal_id: u64, - ) -> Result { - fail_point!(Arc::clone(&self.fp_registry), "runtime-wal-replay", |_| { - Err(RuntimeWalOpenError::Replay(SlateDBError::from( - std::io::Error::other("runtime WAL replay failpoint"), - ))) - }); - let id = SsTableId::Wal(wal_id); - let object_store = self.object_stores.store_for(&id); - let path = self.path(&id); - let mut tag = ObjectStoreCallTag::new(self.kind, SstType::Wal); - let mut validation_retries = 0_usize; - loop { - let object = ReadOnlyObject { - object_store: Arc::clone(&object_store), - path: path.clone(), - tag, - }; - match self.sst_format.read_info_and_version_staged(&object).await { - Ok((info, version)) => return Ok(SsTableHandle::new(id, version, info)), - Err(StagedSstInfoError::Initial(error)) if error.has_object_store_not_found() => { - return Err(RuntimeWalOpenError::MissingInitialObject( - error.with_path(&path), - )); - } - Err(StagedSstInfoError::Initial(error)) | Err(StagedSstInfoError::Later(error)) => { - let error = error.with_path(&path); - let Some(reason) = error.maybe_validation_retry_reason() else { - return Err(RuntimeWalOpenError::Replay(error)); - }; - if validation_retries >= MAX_VALIDATION_RETRIES { - return Err(RuntimeWalOpenError::Replay(error)); - } - validation_retries += 1; - tag.retry = Some(reason); - warn!( - "retrying runtime WAL read after validation failure [wal_id={}, reason={:?}, error={}]", - wal_id, reason, error - ); - } - } - } - } - #[cfg(test)] pub(crate) async fn read_sst_version(&self, id: &SsTableId) -> Result { - let (_, version) = - read_obj!(self, id, |obj| self.sst_format.read_info_and_version(&obj)).await?; + let (_, version) = read_sst_obj!( + self.object_stores.store_for(id), + self.path(id), + ObjectStoreCallTag::new(self.kind, SstType::from(id)), + |obj| self.sst_format.read_info_and_version(&obj) + ) + .await?; Ok(version) } @@ -1294,9 +612,12 @@ impl TableStore { } } } - read_obj!(self, &handle.id, |obj| self - .sst_format - .read_filters(&handle.info, &obj)) + read_sst_obj!( + self.object_stores.store_for(&handle.id), + self.path(&handle.id), + ObjectStoreCallTag::new(self.kind, SstType::from(&handle.id)), + |obj| self.sst_format.read_filters(&handle.info, &obj) + ) .await } @@ -1327,9 +648,12 @@ impl TableStore { return Ok(Some(stats.as_ref().clone())); } } - read_obj!(self, &handle.id, |obj| self - .sst_format - .read_stats(&handle.info, &obj)) + read_sst_obj!( + self.object_stores.store_for(&handle.id), + self.path(&handle.id), + ObjectStoreCallTag::new(self.kind, SstType::from(&handle.id)), + |obj| self.sst_format.read_stats(&handle.info, &obj) + ) .await } @@ -1358,9 +682,12 @@ impl TableStore { return Ok(index); } } - let index = read_obj!(self, &handle.id, |obj| self - .sst_format - .read_index(&handle.info, &obj)) + let index = read_sst_obj!( + self.object_stores.store_for(&handle.id), + self.path(&handle.id), + ObjectStoreCallTag::new(self.kind, SstType::from(&handle.id)), + |obj| self.sst_format.read_index(&handle.info, &obj) + ) .await?; Ok(Arc::new(index)) } @@ -1489,6 +816,57 @@ impl TableStore { .await } + /// Returns the smallest contiguous block range, starting at `first_block` in + /// `order`, whose encoded size is at least `target_bytes`. If the SST boundary + /// is reached first, all remaining blocks in that direction are returned. + pub(crate) fn block_range_for_target_bytes( + &self, + handle: &SsTableHandle, + index: &SsTableIndexOwned, + first_block: usize, + target_bytes: usize, + order: IterationOrder, + ) -> Range { + assert!(target_bytes > 0); + + let index = index.borrow(); + let block_meta = index.block_meta(); + let num_blocks = block_meta.len(); + assert!(first_block < num_blocks); + + let target_bytes = u64::try_from(target_bytes).unwrap_or(u64::MAX); + match order { + IterationOrder::Ascending => { + let mut blocks = first_block..first_block + 1; + loop { + let byte_range = + self.sst_format + .block_range(blocks.clone(), &handle.info, &index); + if byte_range.end.saturating_sub(byte_range.start) >= target_bytes + || blocks.end == num_blocks + { + return blocks; + } + blocks.end += 1; + } + } + IterationOrder::Descending => { + let mut blocks = first_block..first_block + 1; + loop { + let byte_range = + self.sst_format + .block_range(blocks.clone(), &handle.info, &index); + if byte_range.end.saturating_sub(byte_range.start) >= target_bytes + || blocks.start == 0 + { + return blocks; + } + blocks.start -= 1; + } + } + } + } + /// Reads specified blocks from an SSTable using the provided index. /// /// This function attempts to read blocks from the cache if available @@ -1632,12 +1010,17 @@ impl TableStore { handle: &SsTableHandle, block: usize, ) -> Result { - read_obj!(self, &handle.id, |obj| async { - let index = self.sst_format.read_index(&handle.info, &obj).await?; - self.sst_format - .read_block(&handle.info, &index, block, &obj) - .await - }) + read_sst_obj!( + self.object_stores.store_for(&handle.id), + self.path(&handle.id), + ObjectStoreCallTag::new(self.kind, SstType::from(&handle.id)), + |obj| async { + let index = self.sst_format.read_index(&handle.info, &obj).await?; + self.sst_format + .read_block(&handle.info, &index, block, &obj) + .await + } + ) .await } @@ -1654,6 +1037,7 @@ impl TableStore { .estimate_encoded_size_compacted(num_entries, size_entries) } + #[allow(unused)] pub(crate) fn estimate_encoded_size_wal( &self, num_entries: usize, @@ -1728,177 +1112,7 @@ impl TableStore { } } -fn validate_wal_sst_info_layout( - info: &SsTableInfo, - object_len: u64, - metadata_offset: u64, - path: &Path, -) -> Result<(), SlateDBError> { - for (name, offset, len) in [ - ("index", info.index_offset, info.index_len), - ("filter", info.filter_offset, info.filter_len), - ("stats", info.stats_offset, info.stats_len), - ] { - let Some(end) = offset.checked_add(len) else { - return Err(invalid_wal_layout(path, format!("{name} range overflow"))); - }; - if end > metadata_offset { - return Err(invalid_wal_layout( - path, - format!("{name} range {offset}..{end} exceeds metadata offset {metadata_offset}"), - )); - } - } - if metadata_offset >= object_len { - return Err(invalid_wal_layout( - path, - format!("metadata offset {metadata_offset} reaches object length {object_len}"), - )); - } - - let filter_end = info - .filter_offset - .checked_add(info.filter_len) - .ok_or_else(|| invalid_wal_layout(path, "filter range overflow".to_string()))?; - if filter_end != info.index_offset { - return Err(invalid_wal_layout( - path, - format!( - "filter end {filter_end} does not equal index offset {}", - info.index_offset - ), - )); - } - let index_end = info - .index_offset - .checked_add(info.index_len) - .ok_or_else(|| invalid_wal_layout(path, "index range overflow".to_string()))?; - let checksum_size = CHECKSUM_SIZE as u64; - if info.index_len < checksum_size { - return Err(invalid_wal_layout( - path, - format!( - "index length {} is smaller than its checksum", - info.index_len - ), - )); - } - if info.filter_len > 0 && info.filter_len < checksum_size { - return Err(invalid_wal_layout( - path, - format!( - "filter length {} is smaller than its checksum", - info.filter_len - ), - )); - } - if info.stats_len > 0 { - if info.stats_len < checksum_size { - return Err(invalid_wal_layout( - path, - format!( - "stats length {} is smaller than its checksum", - info.stats_len - ), - )); - } - if info.stats_offset != index_end { - return Err(invalid_wal_layout( - path, - format!( - "index end {index_end} does not equal stats offset {}", - info.stats_offset - ), - )); - } - } - let data_end = if info.stats_len > 0 { - info.stats_offset - .checked_add(info.stats_len) - .ok_or_else(|| invalid_wal_layout(path, "stats range overflow".to_string()))? - } else { - index_end - }; - if data_end != metadata_offset { - return Err(invalid_wal_layout( - path, - format!("last WAL section ends at {data_end}, metadata starts at {metadata_offset}"), - )); - } - Ok(()) -} - -fn validate_wal_sst_index_layout( - info: &SsTableInfo, - index: &SsTableIndexOwned, - path: &Path, -) -> Result<(), SlateDBError> { - let block_meta = index.borrow().block_meta(); - if block_meta.is_empty() { - if info.filter_offset != 0 { - return Err(invalid_wal_layout( - path, - format!( - "empty block index has nonzero data length {}", - info.filter_offset - ), - )); - } - return Ok(()); - } - let mut previous_offset = None; - for block in 0..block_meta.len() { - let offset = block_meta.get(block).offset(); - if block == 0 && offset != 0 { - return Err(invalid_wal_layout( - path, - format!("first block offset {offset} is not zero"), - )); - } - if offset >= info.filter_offset { - return Err(invalid_wal_layout( - path, - format!( - "block {block} offset {offset} reaches data end {}", - info.filter_offset - ), - )); - } - if previous_offset.is_some_and(|previous| offset <= previous) { - return Err(invalid_wal_layout( - path, - format!("block {block} offset {offset} is not strictly increasing"), - )); - } - previous_offset = Some(offset); - } - Ok(()) -} - -fn invalid_wal_layout(path: &Path, reason: String) -> SlateDBError { - SlateDBError::WalDataError(Arc::new(std::io::Error::new( - std::io::ErrorKind::InvalidData, - format!("invalid WAL SST layout at {path}: {reason}"), - ))) -} - -fn invalid_wal_size( - path: &Path, - wal_id: u64, - expected_size: Option, - actual_size: u64, -) -> SlateDBError { - SlateDBError::WalDataError(Arc::new(std::io::Error::new( - std::io::ErrorKind::InvalidData, - format!( - "WAL {wal_id} size changed at {path}: expected {}, got {actual_size}", - expected_size - .map(|size| size.to_string()) - .unwrap_or_else(|| "unknown".to_string()) - ), - ))) -} - +#[allow(unused)] async fn wal_object_exists( object_store: &Arc, path: &Path, @@ -1910,45 +1124,6 @@ async fn wal_object_exists( } } -/// Number of additional attempts after an SST read fails validation. The -/// reissue carries a [`RetryReason`](crate::error::RetryReason) so a caching -/// wrapper drops its local copy. -const MAX_VALIDATION_RETRIES: usize = 1; - -/// Runs `read` with the source/type `tag`, reissuing it with a -/// [`RetryReason`](crate::error::RetryReason) set on the tag when the result is -/// a recoverable validation failure. -/// -/// This is done to enable object store wrappers like a cache to know when -/// to drop a cached entry that failed validation and retry the read from the -/// source of truth (object store) instead of repeatedly returning the same -/// invalid cached entry. -async fn read_with_validation_retry( - mut tag: ObjectStoreCallTag, - mut read: impl FnMut(ObjectStoreCallTag) -> Fut, -) -> Result -where - Fut: std::future::Future>, -{ - for _ in 0..MAX_VALIDATION_RETRIES { - let result = read(tag).await; - match result { - Err(ref err) => match err.maybe_validation_retry_reason() { - Some(reason) => { - warn!( - "retrying SST read after validation failure [reason={:?}, error={}]", - reason, err - ); - tag.retry = Some(reason); - } - None => return result, - }, - Ok(_) => return result, - } - } - read(tag).await -} - /// Builds a [`BufWriter`] whose upload carries `tag` in its extensions. fn tagged_buf_writer( object_store: Arc, @@ -2101,10 +1276,9 @@ mod tests { use futures::future; use futures::StreamExt; use object_store::{memory::InMemory, path::Path, ObjectStore, ObjectStoreExt}; - use proptest::prelude::any; - use proptest::proptest; use rstest::rstest; use std::collections::VecDeque; + use std::ops::Range; use std::sync::Arc; use crate::block_cache_policy::BlockCachePolicy; @@ -2112,96 +1286,28 @@ mod tests { use crate::db_cache::CacheTarget; use crate::db_cache::SplitCache; use crate::db_cache::{CachedKey, DbCache, DbCacheWrapper}; + use crate::db_state::{SsTableHandle, SsTableInfo}; use crate::error; + use crate::flatbuffer_types::{ + BlockMeta, BlockMetaArgs, SsTableIndex, SsTableIndexArgs, SsTableIndexOwned, + }; use crate::format::block::Block; - use crate::format::sst::SsTableFormat; + use crate::format::sst::{SsTableFormat, SST_FORMAT_VERSION_LATEST}; + use crate::iter::IterationOrder; use crate::manifest::SsTableView; use crate::object_stores::ObjectStores; use crate::retrying_object_store::RetryingObjectStore; use crate::sst_iter::{SstIterator, SstIteratorOptions}; - use crate::tablestore::{ - validate_wal_sst_index_layout, validate_wal_sst_info_layout, TableStore, TableStoreKind, - }; + use crate::tablestore::{TableStore, TableStoreKind}; use crate::test_utils::FlakyObjectStore; use crate::test_utils::{assert_iterator, build_test_sst}; use crate::types::{RowEntry, ValueDeletable}; - use crate::{ - block_iterator::BlockIteratorLatest, - db_state::{SsTableId, SsTableInfo, SstType}, - iter::RowEntryIterator, - }; + use crate::{block_iterator::BlockIteratorLatest, db_state::SsTableId, iter::RowEntryIterator}; use slatedb_common::clock::DefaultSystemClock; use slatedb_common::DbRand; const ROOT: &str = "/root"; - #[test] - fn should_reject_invalid_wal_section_layouts_before_reading_index_bytes() { - let path = Path::from("/root/wal/00000000000000000001.sst"); - let valid = SsTableInfo { - filter_offset: 16, - filter_len: 4, - index_offset: 20, - index_len: 8, - sst_type: SstType::Wal, - ..SsTableInfo::default() - }; - assert!(validate_wal_sst_info_layout(&valid, 128, 28, &path).is_ok()); - - let invalid = [ - SsTableInfo { - index_offset: u64::MAX, - index_len: 8, - ..valid.clone() - }, - SsTableInfo { - index_offset: 21, - ..valid.clone() - }, - SsTableInfo { - index_len: 3, - ..valid.clone() - }, - SsTableInfo { - filter_len: 3, - index_offset: 19, - ..valid.clone() - }, - SsTableInfo { - stats_offset: 28, - stats_len: 3, - ..valid.clone() - }, - SsTableInfo { - stats_offset: 29, - stats_len: 4, - ..valid.clone() - }, - ]; - for info in invalid { - assert!( - validate_wal_sst_info_layout(&info, 128, 28, &path).is_err(), - "invalid WAL layout was accepted: {info:?}" - ); - } - assert!(validate_wal_sst_info_layout(&valid, 128, 29, &path).is_err()); - } - - #[tokio::test] - async fn should_reject_unindexed_wal_data() { - let path = Path::from("/root/wal/00000000000000000001.sst"); - let encoded = SsTableFormat::default() - .wal_table_builder() - .build() - .await - .unwrap(); - let mut info = encoded.info.clone(); - info.filter_offset = 1; - info.index_offset = 1; - - assert!(validate_wal_sst_index_layout(&info, &encoded.index, &path).is_err()); - } - /// Wraps an object store: counts range-bounded `get_opts` calls and pauses the first /// one until `release` is notified. Other methods just delegate. Shared by the /// concurrent-read dedup tests. @@ -2294,6 +1400,33 @@ mod tests { Arc::new(InMemory::new()) } + fn build_index(offsets: &[u64]) -> SsTableIndexOwned { + let mut builder = flatbuffers::FlatBufferBuilder::new(); + let block_meta = offsets + .iter() + .enumerate() + .map(|(block, offset)| { + let first_key = builder.create_vector(block.to_string().as_bytes()); + BlockMeta::create( + &mut builder, + &BlockMetaArgs { + offset: *offset, + first_key: Some(first_key), + }, + ) + }) + .collect::>(); + let block_meta = builder.create_vector(&block_meta); + let index = SsTableIndex::create( + &mut builder, + &SsTableIndexArgs { + block_meta: Some(block_meta), + }, + ); + builder.finish(index, None); + SsTableIndexOwned::new(Bytes::copy_from_slice(builder.finished_data())).unwrap() + } + async fn count_ssts_in(store: &Arc) -> usize { store .list(None) @@ -3412,7 +2545,7 @@ mod tests { // Create id1, id2, and i3 as three random UUIDs that have been sorted ascending. // Need to do this because the Ulids are sometimes generated in the same millisecond // and the random suffix is used to break the tie, which might be out of order. - let mut ulids = (0..3).map(|_| ulid::Ulid::new()).collect::>(); + let mut ulids = (0..3).map(|_| Ulid::new()).collect::>(); ulids.sort(); let (id1, id2, id3) = ( SsTableId::Compacted(ulids[0]), @@ -3789,20 +2922,91 @@ mod tests { assert_eq!(metadata.location, path); } - proptest! { - #[test] - fn convert_bytes_to_blocks_precise_when_aligned_with_block_size( - block_size in any::(), - num_blocks in any::(), - ) { - let os = Arc::new(InMemory::new()); - let format = SsTableFormat { block_size, ..SsTableFormat::default() }; - let ts = Arc::new(TableStore::new(ObjectStores::new(os, None), - format, Path::from(ROOT), None, TableStoreKind::Main, BlockCachePolicy::default())); - if let Some(bytes) = block_size.checked_mul(num_blocks) { - assert_eq!(num_blocks, ts.bytes_to_blocks(bytes)); - } - } + #[rstest] + #[case::ascending_one_block(IterationOrder::Ascending, 0, 100, 0..1)] + #[case::ascending_crosses_boundary(IterationOrder::Ascending, 0, 101, 0..2)] + #[case::ascending_exact_boundary(IterationOrder::Ascending, 0, 250, 0..2)] + #[case::ascending_exhausts_sst(IterationOrder::Ascending, 1, 1_000, 1..4)] + #[case::descending_one_block(IterationOrder::Descending, 3, 200, 3..4)] + #[case::descending_crosses_boundary(IterationOrder::Descending, 3, 201, 2..4)] + #[case::descending_exact_boundary(IterationOrder::Descending, 3, 250, 2..4)] + #[case::descending_exhausts_sst(IterationOrder::Descending, 2, 1_000, 0..3)] + fn block_range_for_target_bytes_is_minimal( + #[case] order: IterationOrder, + #[case] first_block: usize, + #[case] target_bytes: usize, + #[case] expected: Range, + ) { + let table_store = TableStore::new( + ObjectStores::new(make_store(), None), + SsTableFormat::default(), + Path::from(ROOT), + None, + TableStoreKind::Main, + BlockCachePolicy::default(), + ); + let handle = SsTableHandle::new( + SsTableId::Compacted(ulid::Ulid::new()), + SST_FORMAT_VERSION_LATEST, + SsTableInfo { + index_offset: 500, + filter_offset: 500, + ..SsTableInfo::default() + }, + ); + // Encoded block sizes are 100, 150, 50, and 200 bytes. + let index = build_index(&[0, 100, 250, 300]); + + let actual = table_store.block_range_for_target_bytes( + &handle, + &index, + first_block, + target_bytes, + order, + ); + + assert_eq!(actual, expected); + } + + #[rstest] + #[case::filter_precedes_index(1, 500, 700)] + #[case::index_follows_data_without_filter(0, 700, 500)] + fn block_range_for_target_bytes_uses_format_for_last_block_end( + #[case] filter_len: u64, + #[case] filter_offset: u64, + #[case] index_offset: u64, + ) { + let table_store = TableStore::new( + ObjectStores::new(make_store(), None), + SsTableFormat::default(), + Path::from(ROOT), + None, + TableStoreKind::Main, + BlockCachePolicy::default(), + ); + let handle = SsTableHandle::new( + SsTableId::Compacted(ulid::Ulid::new()), + SST_FORMAT_VERSION_LATEST, + SsTableInfo { + index_offset, + filter_offset, + filter_len, + ..SsTableInfo::default() + }, + ); + let index = build_index(&[0, 100, 250, 300]); + + let actual = table_store.block_range_for_target_bytes( + &handle, + &index, + 3, + 201, + IterationOrder::Descending, + ); + + // The last block ends at byte 500 and is 200 bytes, so one preceding + // block is required to meet a 201-byte target. + assert_eq!(actual, 2..4); } /// End-to-end test: concurrent index reads through `TableStore` issue a single @@ -4048,21 +3252,12 @@ mod tests { use crate::error::{RetryReason, SlateDBError}; use crate::format::sst::SsTableFormat; use crate::object_stores::ObjectStores; - use crate::replay_task_scope::ReplayTaskScope; use crate::tablestore::TableStore; use crate::test_utils::{build_test_sst, RecordingObjectStore}; use object_store::memory::InMemory; - use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering}; + use std::sync::atomic::{AtomicUsize, Ordering}; use std::sync::{Arc, Mutex}; - struct RetryCancellationProbe(Arc); - - impl Drop for RetryCancellationProbe { - fn drop(&mut self) { - self.0.store(true, Ordering::Release); - } - } - fn format() -> SsTableFormat { SsTableFormat { block_size: 32, @@ -4244,48 +3439,6 @@ mod tests { ); } - #[tokio::test] - async fn validation_retry_is_cancelled_when_its_owner_is_aborted() { - let scope = ReplayTaskScope::new(); - let attempts = Arc::new(AtomicUsize::new(0)); - let retry_started = Arc::new(tokio::sync::Notify::new()); - let cancelled = Arc::new(AtomicBool::new(false)); - let task_attempts = Arc::clone(&attempts); - let task_retry_started = Arc::clone(&retry_started); - let task_cancelled = Arc::clone(&cancelled); - let task = scope.spawn(async move { - read_with_validation_retry( - ObjectStoreCallTag::new(TableStoreKind::Reader, SstType::Wal), - move |_| { - let attempt = task_attempts.fetch_add(1, Ordering::SeqCst); - let retry_started = Arc::clone(&task_retry_started); - let cancelled = Arc::clone(&task_cancelled); - async move { - if attempt == 0 { - return Err(SlateDBError::ChecksumMismatch { path: None }); - } - let _probe = RetryCancellationProbe(cancelled); - retry_started.notify_one(); - std::future::pending::>().await - } - }, - ) - .await - }); - tokio::time::timeout(std::time::Duration::from_secs(1), retry_started.notified()) - .await - .expect("validation retry did not start"); - - task.abort(); - let _ = task.await; - scope.shutdown().await; - assert_eq!(attempts.load(Ordering::SeqCst), 2); - assert!( - cancelled.load(Ordering::Acquire), - "validation retry future outlived its owner" - ); - } - #[tokio::test] async fn compacted_writes_carry_source_and_compacted_type() { let (recording, ts) = recording_store(TableStoreKind::Compactor); diff --git a/slatedb/src/test_utils.rs b/slatedb/src/test_utils.rs index 9842c546d1..feef8d5f7b 100644 --- a/slatedb/src/test_utils.rs +++ b/slatedb/src/test_utils.rs @@ -368,11 +368,11 @@ where pub(crate) async fn seed_database( db: &Db, table: &BTreeMap, - await_durable: bool, + wait_for_durability: bool, ) -> Result<(), crate::Error> { let put_options = PutOptions::default(); let write_options = WriteOptions { - await_durable, + await_durable: wait_for_durability, ..Default::default() }; @@ -461,10 +461,7 @@ pub(crate) async fn build_sorted_runs( let ssts = write_ssts(table_store, entries, max_sst_size).await; sr_ssts.extend(ssts.into_iter().map(SsTableView::identity)); } - sorted_runs.push(SortedRun { - id: sr_id as u32, - sst_views: sr_ssts, - }); + sorted_runs.push(SortedRun::new(sr_id as u32, sr_ssts)); } sorted_runs @@ -571,6 +568,8 @@ pub(crate) struct FlakyObjectStore { // get_range: truncate response body to this many bytes on first N attempts (0 = no truncation) truncate_get_range_bytes: AtomicUsize, truncate_get_range_count: AtomicUsize, + // Get: return NotFound on the next N GETs + fail_first_get_not_found: AtomicUsize, } impl FlakyObjectStore { @@ -596,9 +595,14 @@ impl FlakyObjectStore { get_range_attempts: AtomicUsize::new(0), truncate_get_range_bytes: AtomicUsize::new(0), truncate_get_range_count: AtomicUsize::new(0), + fail_first_get_not_found: AtomicUsize::new(0), } } + pub(crate) fn with_get_not_found_failures(&self, n: usize) { + self.fail_first_get_not_found.store(n, Ordering::SeqCst); + } + pub(crate) fn with_put_precondition_always(self) -> Self { self.put_precondition_always.store(true, Ordering::SeqCst); self @@ -723,7 +727,26 @@ impl ObjectStore for FlakyObjectStore { &self, location: &Path, options: GetOptions, - ) -> object_store::Result { + ) -> object_store::Result { + if self + .fail_first_get_not_found + .fetch_update(Ordering::SeqCst, Ordering::SeqCst, |v| { + if v > 0 { + Some(v - 1) + } else { + None + } + }) + .is_ok() + { + return Err(object_store::Error::NotFound { + path: location.to_string(), + source: Box::new(std::io::Error::new( + std::io::ErrorKind::NotFound, + "injected not-found (deleted between LIST and GET)", + )), + }); + } if options.head { self.head_attempts.fetch_add(1, Ordering::SeqCst); if self @@ -786,9 +809,9 @@ impl ObjectStore for FlakyObjectStore { let extensions = result.extensions.clone(); let body = result.bytes().await?; let truncated = body.slice(..truncate_bytes.min(body.len())); - return Ok(object_store::GetResult { + return Ok(GetResult { payload: object_store::GetResultPayload::Stream( - futures::stream::once(async { Ok(truncated) }).boxed(), + stream::once(async { Ok(truncated) }).boxed(), ), meta, range, @@ -866,7 +889,7 @@ impl ObjectStore for FlakyObjectStore { async fn put_multipart_opts( &self, location: &Path, - opts: object_store::PutMultipartOptions, + opts: PutMultipartOptions, ) -> object_store::Result> { self.put_multipart_attempts.fetch_add(1, Ordering::SeqCst); self.inner.put_multipart_opts(location, opts).await @@ -1150,7 +1173,7 @@ impl ObjectStore for GatedObjectStore { &self, location: &Path, options: GetOptions, - ) -> object_store::Result { + ) -> object_store::Result { if options.head { self.head_gate.wait().await?; } else { @@ -1172,7 +1195,7 @@ impl ObjectStore for GatedObjectStore { async fn put_multipart_opts( &self, location: &Path, - opts: object_store::PutMultipartOptions, + opts: PutMultipartOptions, ) -> object_store::Result> { self.put_multipart_opts_gate.wait().await?; self.inner.put_multipart_opts(location, opts).await @@ -1246,7 +1269,7 @@ impl ObjectStore for GatedObjectStore { &self, from: &Path, to: &Path, - options: object_store::RenameOptions, + options: RenameOptions, ) -> object_store::Result<()> { self.rename_gate.wait().await?; self.inner.rename_opts(from, to, options).await @@ -1417,6 +1440,35 @@ impl crate::prefix_extractor::PrefixExtractor for FixedThreeBytePrefixExtractor } } +/// Test extractor that segments on the leading `data` / `idx` path +/// component, modelling a store that keeps bulky records in one segment and +/// a smaller index over them in another. Tenants live in the *next* +/// component (`data/{tenant}/…`), so a tenant's rows are split across both +/// segments rather than forming one contiguous key range. +// Only the union tests in `clone.rs` use this, and those need `wal_disable`. +#[cfg(feature = "wal_disable")] +#[derive(Debug)] +pub(crate) struct DataIdxPrefixExtractor; + +#[cfg(feature = "wal_disable")] +impl crate::prefix_extractor::PrefixExtractor for DataIdxPrefixExtractor { + fn name(&self) -> &str { + "data-idx" + } + fn prefix_len(&self, target: &crate::prefix_extractor::PrefixTarget) -> Option { + let key: &[u8] = match target { + crate::prefix_extractor::PrefixTarget::Point(b) + | crate::prefix_extractor::PrefixTarget::Prefix(b) => b.as_ref(), + }; + for kind in [b"data".as_slice(), b"idx".as_slice()] { + if key == kind || key.starts_with(&[kind, b"/".as_slice()].concat()) { + return Some(kind.len()); + } + } + None + } +} + /// Test extractor that deliberately violates the /// [`crate::prefix_extractor::PrefixExtractor`] `Point` invariant by /// returning different prefix lengths for keys that share a common @@ -1574,6 +1626,7 @@ mod tests { pub(crate) enum RecordedCall { Get { head: bool, + #[allow(dead_code)] range_bytes: Option, kind: Option, sst_type: Option, @@ -1647,6 +1700,7 @@ impl RecordingObjectStore { .collect() } + #[allow(dead_code)] pub(crate) fn get_range_sizes(&self) -> Vec { self.calls .lock() @@ -1661,6 +1715,7 @@ impl RecordingObjectStore { .collect() } + #[allow(dead_code)] pub(crate) fn list_calls(&self) -> usize { self.list_calls.load(Ordering::SeqCst) } @@ -1703,7 +1758,7 @@ impl ObjectStore for RecordingObjectStore { &self, location: &Path, options: GetOptions, - ) -> object_store::Result { + ) -> object_store::Result { let tag = ObjectStoreCallTag::from_extensions(&options.extensions); self.calls.lock().push(RecordedCall::Get { head: options.head, @@ -1735,7 +1790,7 @@ impl ObjectStore for RecordingObjectStore { async fn put_multipart_opts( &self, location: &Path, - opts: object_store::PutMultipartOptions, + opts: PutMultipartOptions, ) -> object_store::Result> { let tag = ObjectStoreCallTag::from_extensions(&opts.extensions); self.calls.lock().push(RecordedCall::PutMultipart { diff --git a/slatedb/src/transaction_manager.rs b/slatedb/src/transaction_manager.rs index 4a67309ec5..f4fa827f5c 100644 --- a/slatedb/src/transaction_manager.rs +++ b/slatedb/src/transaction_manager.rs @@ -1902,7 +1902,7 @@ mod tests { // For every active transaction, verify the smoking-gun invariant: // rw_conflict <=> exists culprit in recent_committed_txns that satisfies rules let inner = txn_manager.inner.read(); - for (_id, txn) in inner.active_txns.iter() { + for txn in inner.active_txns.values() { // Only meaningful for SSI; read-only transactions can also have reads, so include them. let rw_conflict = inner.has_read_write_conflict( &txn.read_keys, diff --git a/slatedb/src/types.rs b/slatedb/src/types.rs index d4306b7b76..8eddcdd328 100644 --- a/slatedb/src/types.rs +++ b/slatedb/src/types.rs @@ -48,13 +48,13 @@ impl RowEntry { pub(crate) fn estimated_size(&self) -> usize { let mut size = self.key.len() + self.value.len(); // Add size for sequence number - size += std::mem::size_of::(); + size += size_of::(); // Add size for timestamps if self.create_ts.is_some() { - size += std::mem::size_of::(); + size += size_of::(); } if self.expire_ts.is_some() { - size += std::mem::size_of::(); + size += size_of::(); } size } @@ -63,20 +63,20 @@ impl RowEntry { /// The `key_prefix_len` is the number of bytes shared with the block's first key. pub(crate) fn encoded_size(&self, key_prefix_len: usize) -> usize { let key_suffix_len = self.key.len() - key_prefix_len; - let mut size = std::mem::size_of::() // key_prefix_len - + std::mem::size_of::() // key_suffix_len + let mut size = size_of::() // key_prefix_len + + size_of::() // key_suffix_len + key_suffix_len - + std::mem::size_of::() // seq - + std::mem::size_of::(); // flags + + size_of::() // seq + + size_of::(); // flags if self.expire_ts.is_some() { - size += std::mem::size_of::(); + size += size_of::(); } if self.create_ts.is_some() { - size += std::mem::size_of::(); + size += size_of::(); } if !self.value.is_tombstone() { - size += std::mem::size_of::(); // value_len + size += size_of::(); // value_len size += self.value.len(); } size diff --git a/slatedb/src/utils.rs b/slatedb/src/utils.rs index 3f5b90d887..85e06c976e 100644 --- a/slatedb/src/utils.rs +++ b/slatedb/src/utils.rs @@ -390,7 +390,7 @@ pub(crate) fn sign_extend(val: u32, bits: u8) -> i32 { /// Returns: /// - The effective max parallelism. pub(crate) fn compute_max_parallel(l0_count: usize, srs: &[SortedRun], cap: usize) -> usize { - let total_ssts = l0_count + srs.iter().map(|sr| sr.sst_views.len()).sum::(); + let total_ssts = l0_count + srs.iter().map(|sr| sr.sst_views().len()).sum::(); total_ssts.min(cap).max(1) } @@ -416,7 +416,7 @@ pub(crate) fn estimate_bytes_before_key(sorted_runs: &[SortedRun], key: &Bytes) return 0; }; sorted_run - .sst_views + .sst_views() .iter() .take(idx) .map(|sst| sst.estimate_size()) @@ -454,7 +454,7 @@ where I::Item: Send, T: Send, F: Fn(I::Item) -> Fut + Send, - Fut: std::future::Future, SlateDBError>> + Send, + Fut: Future, SlateDBError>> + Send, { let mut out = VecDeque::new(); @@ -522,11 +522,8 @@ pub(crate) fn panic_string(panic: &Box) -> String { /// - (Err(SlateDBError::BackgroundTaskPanic), Some(payload)) if the task panicked pub(crate) fn split_unwind_result( name: String, - unwind_result: Result, Box>, -) -> ( - Result<(), SlateDBError>, - Option>, -) { + unwind_result: Result, Box>, +) -> (Result<(), SlateDBError>, Option>) { match unwind_result { Ok(result) => (result, None), Err(payload) => (Err(SlateDBError::BackgroundTaskPanic(name)), Some(payload)), @@ -552,10 +549,7 @@ pub(crate) fn split_unwind_result( pub(crate) fn split_join_result( name: String, join_result: Result, tokio::task::JoinError>, -) -> ( - Result<(), SlateDBError>, - Option>, -) { +) -> (Result<(), SlateDBError>, Option>) { match join_result { Ok(task_result) => (task_result, None), Err(join_error) => { @@ -1356,19 +1350,19 @@ mod tests { #[test] fn test_estimate_bytes_before_key() { - let run1 = SortedRun { - id: 1, - sst_views: vec![ + let run1 = SortedRun::new( + 1, + [ make_sst_view("a", 10), make_sst_view("k", 20), // k < m < z, so only "a" counts make_sst_view("z", 30), ], - }; - let run2 = SortedRun { - id: 2, + ); + let run2 = SortedRun::new( + 2, // f < m < ..., so only "b" counts - sst_views: vec![make_sst_view("b", 40), make_sst_view("f", 50)], - }; + [make_sst_view("b", 40), make_sst_view("f", 50)], + ); let key = Bytes::from("m"); let total = estimate_bytes_before_key(&[run1, run2], &key); @@ -1513,8 +1507,7 @@ mod tests { #[test] fn test_split_unwind_result_ok_ok() { // Given: a successful unwind result - let unwind_result: Result, Box> = - Ok(Ok(())); + let unwind_result: Result, Box> = Ok(Ok(())); // When: we split the result let (result, payload) = super::split_unwind_result("test".to_string(), unwind_result); @@ -1527,7 +1520,7 @@ mod tests { #[test] fn test_split_unwind_result_ok_error() { // Given: an unwind result with a task error - let unwind_result: Result, Box> = + let unwind_result: Result, Box> = Ok(Err(SlateDBError::Fenced)); // When: we split the result @@ -1542,7 +1535,7 @@ mod tests { fn test_split_unwind_result_panic() { // Given: an unwind result that panicked with a non-SlateDBError (e.g., a string) let panic_msg = "something went wrong"; - let unwind_result: Result, Box> = + let unwind_result: Result, Box> = Err(Box::new(panic_msg)); // When: we split the result diff --git a/slatedb/src/wal/mod.rs b/slatedb/src/wal/mod.rs index e709227412..daf7d93bc7 100644 --- a/slatedb/src/wal/mod.rs +++ b/slatedb/src/wal/mod.rs @@ -3,19 +3,25 @@ use crate::manifest::store::FenceableManifest; use crate::{CloseReason, ErrorKind, RowEntry, VersionedManifest}; use async_trait::async_trait; use futures::future::BoxFuture; +use object_store::path::Path; use std::error::Error; use std::fmt::{Display, Formatter}; -use std::ops::{Bound, Range}; +use std::ops::{Bound, Range, RangeFrom}; use std::sync::Arc; +use std::time::Duration; +pub(crate) mod slatedb; #[cfg(test)] pub(crate) mod test_utils; pub(crate) mod wal_disabled; -pub(crate) mod wal_sst_builder; -pub(crate) mod writer_init; + +pub use crate::wal::slatedb::reader::{ + SlateDbWalReader, SlateDbWalReaderBuilder, SlateDbWalReaderOptions, +}; /// A range of WAL File IDs -pub struct WalFileRange(Bound, Bound); +#[derive(Clone, Debug, Eq, PartialEq)] +pub struct WalFileRange(pub Bound, pub Bound); impl From> for WalFileRange { fn from(range: Range) -> Self { @@ -23,6 +29,12 @@ impl From> for WalFileRange { } } +impl From> for WalFileRange { + fn from(range: RangeFrom) -> Self { + WalFileRange(Bound::Included(range.start), Bound::Unbounded) + } +} + impl TryFrom for Range { type Error = (); @@ -41,7 +53,7 @@ pub enum WalError { /// The WAL writer was fenced Fenced, /// A WalIterator observed that the tail of the WAL was truncated while iterating. - WalTruncated, + WalTruncated(u64), /// Operation against wal after it was closed Closed, /// WAL is unavailable, e.g. due to an I/O error or error in the backing storage system @@ -56,7 +68,7 @@ impl Display for WalError { fn fmt(&self, f: &mut Formatter<'_>) -> std::fmt::Result { match self { WalError::Fenced => write!(f, "WAL writer was fenced"), - WalError::WalTruncated => write!(f, "WAL was truncated"), + WalError::WalTruncated(wal_id) => write!(f, "WAL was truncated at file {}", *wal_id), WalError::Closed => write!(f, "WAL is closed"), WalError::Unavailable(source) => write!(f, "WAL is unavailable: {source}"), WalError::DataError(source) => write!(f, "WAL data error: {source}"), @@ -113,10 +125,9 @@ impl WriterManifest { /// The result returned by [`WriterInit::fence_and_init`] pub struct WriterInitResult { - // TODO: change me to an iterator /// An iterator that returns writes that must be replayed before starting SlateDB to recover /// data from the WAL. - pub replay_range: WalFileRange, + pub replay_iterator: Box, /// The WAL writer that will be used to append new writes to the WAL pub wal_writer: Box, } @@ -143,7 +154,7 @@ pub struct WriterInitResult { /// rows in WAL files between [`WriterManifest::replay_after_wal_id`] (exclusive) and the /// current end of the WAL. #[async_trait] -pub trait WriterInit { +pub trait WriterInit: Send + Sync + 'static { /// Fences the WAL and returns a [`WriterInitResult`] with a [`WalWriter`] and /// [`WalReplayIterator`] used to recover writes that have not yet been flushed to the tree. async fn fence_and_init( @@ -174,7 +185,7 @@ pub struct WalStatus { #[derive(Debug, Clone)] pub enum WalEvent { /// Emitted when a WAL file is durably flushed to storage. On receipt of this event, SlateDB - /// notifies write tasks blocked on [`crate::config::WriteOptions::await_durable`] + /// advances the durable sequence number and notifies durability waiters. WalFlushed(WalStatus), /// Emitted when the WAL has closed with the final wal status containing the closed reason WalClosed(WalStatus), @@ -217,6 +228,18 @@ pub trait WalWriter: Send { /// future that receives the result of the flush once it completes. async fn flush(&mut self) -> Result; + /// Returns true if the WAL implementation wants to request that the current in-memory + /// writes be flushed to a new l0. WAL implementations can use this to (1) bound the range + /// of writes that need to be replayed when SlateDB restarts, and (2) push data to L0s earlier + /// so that it's available to readers, which poll the latest manifest. + /// + /// ## Arguments + /// - `replay_after_wal_id`: The WAL ID used as the replay point for the last memtable + /// that was flushed to L0 + fn should_flush_memtable(&self, _replay_after_wal_id: u64) -> bool { + false + } + /// Returns a `WalObserver` for reading [`WalStatus`] and subscribing to events. fn observer(&self) -> Box; @@ -230,6 +253,127 @@ pub trait WalWriter: Send { async fn close(&mut self) -> Result<(), WalError>; } +/// Rows returned by [`WalIterator`] +#[derive(Clone)] +pub struct WalRows { + /// The rows read from the WAL File. All the rows with a given sequence number must be present + /// in th same [`WalRows`]. + pub rows: Vec, + /// The id of the last WAL File for which all rows have been consumed by the iterator and + /// returned wither in this [`WalRows`] or a [`WalRows`] returned by an earlier call to + /// [`WalIterator::next`] + pub last_consumed_wal_file_id: u64, +} + +/// An iterator over rows in some range of the WAL +#[async_trait] +pub trait WalIterator: Send + 'static { + /// Returns the next set of rows. Rows must be returned in sequence and WAL File order. + /// Returns None when iterator's range is exhausted. Iterators created using an unbounded + /// end range that have exhausted the current WAL block until new rows are appended and never + /// return `None`. + /// Returns [`WalError::WalTruncated`] if the iterator observes that the WAL was truncated + /// while iterating. + async fn next(&mut self) -> Result, WalError>; +} + +/// API for reading from the WAL. Used by the Reader/ +#[async_trait] +pub trait WalReader: Send + Sync + 'static { + /// Returns an iterator over the specified range of WAL File IDs. The start of the range must + /// not be `Unbounded`. If the end of the range is `Unbounded` then the returned iterator + /// continues returning writes as new writes are appended to the WAL. Otherwise, it returns + /// `None` upon reaching the end of the range. + async fn iterator( + &self, + wal_file_id_range: WalFileRange, + ) -> Result, WalError>; + + /// Returns the ID of the last WAL file currently present after `replay_after_wal_id`, or + /// `replay_after_wal_id` if no later WAL file is present. Implementations may use + /// `replay_after_wal_id` as a known lower bound when locating the end of the WAL. + async fn last_wal_file_id(&self, replay_after_wal_id: u64) -> Result; +} + +/// Trait that defines the contract between SlateDB's garbage collector and a custom WAL +/// implementation. SlateDB tracks the set of currently referenced WAL ranges in its manifest. +/// When the Garbage Collector runs, it computes this set and calls [`WalGc::collect`] so that +/// the implementation can clean up any un-referenced WAL storage. +#[async_trait] +pub trait WalGc: Send + Sync + 'static { + /// Hook for garbage collecting the WAL. Takes a list of ranges of WAL Files that are currently + /// referenced by some active Manifest. The implementation may delete any WAL File that is not + /// included in the ranges in this list. + async fn collect( + &self, + referenced_ranges: Vec, + min_age: Duration, + dry_run: bool, + ) -> Result<(), WalError>; +} + +/// Administrative operations for a WAL implementation. +#[async_trait] +pub trait WalAdmin: Send + Sync + 'static { + /// Creates a garbage collector scoped to the WAL at `path`. + /// + /// ## Arguments + /// - `path`: The database path whose WAL should be garbage collected. + /// + /// ## Returns + /// A garbage collector that can remove unreferenced WAL files at `path`. + fn garbage_collector(&self, path: &Path) -> Arc; + + /// Deletes the WAL at `path`. + /// + /// ## Arguments + /// - `path`: The database path whose WAL should be deleted. + /// - `dry_run`: If set to true, the implementation should just return the list of resources + /// that would be deleted without actually deleting anything. + /// + /// ## Returns + /// `Ok(resources)` after the WAL has been deleted, or a [`WalError`] if deletion fails, where + /// `resources` is a list of descriptions of resources that were deleted by this fn. This + /// list is used to display the output of deleting the WAL (e.g. in logs or tool output) + async fn delete_wal(&self, path: &Path, dry_run: bool) -> Result, WalError>; + + /// Given a path and WAL ID range, returns true if the WAL at that path is empty within the + /// specified range. A WAL is empty if it holds no records. + /// + /// ## Arguments + /// - `path`: The database path containing the WAL. + /// - `replay_after_wal_id`: The exclusive lower bound of the WAL range to inspect. + /// - `wal_id_last_seen`: The inclusive upper bound of the WAL range to inspect. + /// + /// ## Returns + /// `Ok(true)` if the referenced WAL contains no records, `Ok(false)` if it contains records, + /// or a [`WalError`] if the WAL could not be inspected. + async fn is_empty( + &self, + path: &Path, + replay_after_wal_id: u64, + wal_id_last_seen: u64, + ) -> Result; + + /// Given a source path and manifest, copy the referenced WAL to a destination path and return + /// a replay range. This call must be idempotent (TODO: clarify) + /// + /// ## Arguments + /// - `from_path`: The db path that holds the source WAL range to be copied + /// - `from_manifest`: The source manifest that identifies the WAL to copy + /// - `to_path`: The db path of the clone that the WAL is being copied to. + /// + /// ## Returns + /// A (u64, u64) pair. The first item will be used as the replay start point (exclusive). The + /// second item should be the id of the last WAL file id in the copied WAL. + async fn clone_wal( + &self, + from_path: &Path, + from_manifest: VersionedManifest, + to_path: &Path, + ) -> Result<(u64, u64), WalError>; +} + impl From for WalError { fn from(status: WalStatus) -> Self { status @@ -246,18 +390,16 @@ impl From for SlateDBError { impl From for WalError { fn from(value: SlateDBError) -> Self { - { - let public: crate::Error = value.clone().into(); - match public.kind() { - ErrorKind::Closed(CloseReason::Fenced) => WalError::Fenced, - ErrorKind::Closed(CloseReason::Clean) => WalError::Closed, - ErrorKind::Closed(_) => WalError::InternalError(Arc::new(value)), - ErrorKind::Unavailable => WalError::Unavailable(Arc::new(value)), - ErrorKind::Invalid => WalError::InternalError(Arc::new(value)), - ErrorKind::Data => WalError::DataError(Arc::new(value)), - ErrorKind::Internal => WalError::InternalError(Arc::new(value)), - ErrorKind::Transaction => WalError::InternalError(Arc::new(value)), - } + let public: crate::Error = value.clone().into(); + match public.kind() { + ErrorKind::Closed(CloseReason::Fenced) => WalError::Fenced, + ErrorKind::Closed(CloseReason::Clean) => WalError::Closed, + ErrorKind::Closed(_) => WalError::InternalError(Arc::new(value)), + ErrorKind::Unavailable => WalError::Unavailable(Arc::new(value)), + ErrorKind::Invalid => WalError::InternalError(Arc::new(value)), + ErrorKind::Data => WalError::DataError(Arc::new(value)), + ErrorKind::Internal => WalError::InternalError(Arc::new(value)), + ErrorKind::Transaction => WalError::InternalError(Arc::new(value)), } } } @@ -266,7 +408,7 @@ impl From for SlateDBError { fn from(value: WalError) -> Self { match value { WalError::Fenced => SlateDBError::Fenced, - WalError::WalTruncated => SlateDBError::WalTruncated, + WalError::WalTruncated(wal_id) => SlateDBError::WalTruncated(wal_id), WalError::Closed => SlateDBError::Closed, WalError::Unavailable(err) => SlateDBError::WalUnavailable(err), WalError::DataError(err) => SlateDBError::WalDataError(err), @@ -288,7 +430,10 @@ mod tests { }; assert_eq!(WalError::Fenced.to_string(), "WAL writer was fenced"); - assert_eq!(WalError::WalTruncated.to_string(), "WAL was truncated"); + assert_eq!( + WalError::WalTruncated(123).to_string(), + "WAL was truncated at file 123" + ); assert_eq!(WalError::Closed.to_string(), "WAL is closed"); assert_eq!( WalError::Unavailable(source()).to_string(), diff --git a/slatedb/src/wal/slatedb/admin.rs b/slatedb/src/wal/slatedb/admin.rs new file mode 100644 index 0000000000..7d586f144a --- /dev/null +++ b/slatedb/src/wal/slatedb/admin.rs @@ -0,0 +1,211 @@ +use crate::block_cache_policy::BlockCachePolicy; +use crate::db_state::SsTableId; +use crate::format::sst::SsTableFormat; +use crate::garbage_collector::stats::GcStats; +use crate::object_stores::ObjectStores; +use crate::paths::PathResolver; +use crate::tablestore::{TableStore, TableStoreKind}; +use crate::wal::slatedb::gc::{SlateDbWalGc, WalGcMode}; +use crate::wal::{WalAdmin, WalError, WalGc}; +use crate::VersionedManifest; +use async_trait::async_trait; +use fail_parallel::{fail_point, FailPointRegistry}; +use futures::StreamExt; +use object_store::path::Path; +use object_store::{ObjectStore, ObjectStoreExt}; +use slatedb_common::clock::DefaultSystemClock; +use slatedb_common::metrics::MetricsRecorderHelper; +use std::sync::Arc; + +#[derive(Clone)] +pub(crate) struct SlateDbWalAdmin { + object_store: Arc, + #[cfg_attr(not(test), allow(dead_code))] + fp_registry: Arc, +} + +impl SlateDbWalAdmin { + pub(crate) fn new( + object_store: Arc, + fp_registry: Arc, + ) -> Self { + Self { + object_store, + fp_registry, + } + } + + fn replay_range(manifest: &VersionedManifest) -> Result<(u64, u64), WalError> { + let replay_after_wal_id = manifest.replay_after_wal_id(); + let wal_id_last_seen = manifest + .next_wal_sst_id() + .checked_sub(1) + .ok_or_else(Self::invalid_manifest)?; + Ok((replay_after_wal_id, wal_id_last_seen)) + } + + fn invalid_manifest() -> WalError { + WalError::InternalError(Arc::new(std::io::Error::new( + std::io::ErrorKind::InvalidInput, + "source manifest must have a positive next WAL file ID", + ))) + } + + fn has_wal_file_ids(replay_after_wal_id: u64, wal_id_last_seen: u64) -> bool { + wal_id_last_seen > replay_after_wal_id + } + + async fn paths_under(&self, path: &Path) -> Result, WalError> { + let mut objects = self.object_store.list(Some(path)); + let mut paths = Vec::new(); + while let Some(object) = objects.next().await { + let object = object.map_err(|err| WalError::Unavailable(Arc::new(err)))?; + paths.push(object.location); + } + Ok(paths) + } +} + +#[async_trait] +impl WalAdmin for SlateDbWalAdmin { + fn garbage_collector(&self, path: &Path) -> Arc { + let table_store = Arc::new(TableStore::new( + ObjectStores::new(self.object_store.clone(), None), + SsTableFormat::default(), + path.clone(), + None, + TableStoreKind::GC, + BlockCachePolicy::default(), + )); + Arc::new(SlateDbWalGc::new( + table_store, + Arc::new(GcStats::new(&MetricsRecorderHelper::noop())), + WalGcMode::Regular, + None, + Arc::new(DefaultSystemClock::new()), + )) + } + + async fn delete_wal(&self, path: &Path, dry_run: bool) -> Result, WalError> { + // Collect the paths first so listing is complete before objects are removed. + let wal_path = PathResolver::from_root(path.clone()).wal_path(); + let paths = self.paths_under(&wal_path).await?; + if !dry_run { + for object_path in &paths { + self.object_store + .delete(object_path) + .await + .map_err(|err| WalError::Unavailable(Arc::new(err)))?; + } + } + Ok(paths.iter().map(Path::to_string).collect()) + } + + async fn is_empty( + &self, + path: &Path, + replay_after_wal_id: u64, + wal_id_last_seen: u64, + ) -> Result { + // Avoid object-store requests when the manifest's WAL range contains no file IDs. + if !Self::has_wal_file_ids(replay_after_wal_id, wal_id_last_seen) { + return Ok(true); + } + + let path_resolver = PathResolver::from_root(path.clone()); + for wal_id in (replay_after_wal_id + 1)..=wal_id_last_seen { + let path = path_resolver.sst_path(&SsTableId::Wal(wal_id)); + let metadata = self + .object_store + .head(&path) + .await + .map_err(|err| WalError::Unavailable(Arc::new(err)))?; + + // Native SlateDB WAL fences are zero-byte WAL objects and contain no records. + if metadata.size > 0 { + return Ok(false); + } + } + Ok(true) + } + + async fn clone_wal( + &self, + from_path: &Path, + from_manifest: VersionedManifest, + to_path: &Path, + ) -> Result<(u64, u64), WalError> { + let (replay_after_wal_id, wal_id_last_seen) = Self::replay_range(&from_manifest)?; + let from_path_resolver = PathResolver::from_root(from_path.clone()); + let to_path_resolver = PathResolver::from_root(to_path.clone()); + + if Self::has_wal_file_ids(replay_after_wal_id, wal_id_last_seen) { + for wal_id in (replay_after_wal_id + 1)..=wal_id_last_seen { + fail_point!(self.fp_registry.clone(), "copy-wal-ssts-io-error", |_| Err( + WalError::Unavailable(Arc::new(std::io::Error::other("oops"))) + )); + + let id = SsTableId::Wal(wal_id); + let source = from_path_resolver.sst_path(&id); + let destination = to_path_resolver.sst_path(&id); + self.object_store + .as_ref() + .copy(&source, &destination) + .await + .map_err(|err| WalError::Unavailable(Arc::new(err)))?; + } + } + + Ok((replay_after_wal_id, wal_id_last_seen)) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use bytes::Bytes; + use object_store::memory::InMemory; + + #[tokio::test] + async fn delete_wal_deletes_only_objects_under_the_wal_prefix() { + let object_store: Arc = Arc::new(InMemory::new()); + let wal_admin = + SlateDbWalAdmin::new(object_store.clone(), Arc::new(FailPointRegistry::new())); + let db_path = Path::from("db"); + let wal_object = PathResolver::from_root(db_path.clone()).sst_path(&SsTableId::Wal(1)); + let non_wal_object = db_path + .clone() + .join("manifest") + .join("00000000000000000001"); + let sibling_object = Path::from("other/wal/00000000000000000002.sst"); + object_store + .put(&wal_object, Bytes::from_static(b"wal").into()) + .await + .unwrap(); + object_store + .put(&non_wal_object, Bytes::from_static(b"keep").into()) + .await + .unwrap(); + object_store + .put(&sibling_object, Bytes::from_static(b"keep").into()) + .await + .unwrap(); + + wal_admin.delete_wal(&db_path, false).await.unwrap(); + + assert!(matches!( + object_store.head(&wal_object).await, + Err(object_store::Error::NotFound { .. }) + )); + assert!(object_store.head(&non_wal_object).await.is_ok()); + assert!(object_store.head(&sibling_object).await.is_ok()); + } + + #[test] + fn creates_a_path_scoped_garbage_collector() { + let object_store: Arc = Arc::new(InMemory::new()); + let wal_admin = SlateDbWalAdmin::new(object_store, Arc::new(FailPointRegistry::new())); + + let _collector = wal_admin.garbage_collector(&Path::from("db")); + } +} diff --git a/slatedb/src/wal/slatedb/gc.rs b/slatedb/src/wal/slatedb/gc.rs new file mode 100644 index 0000000000..baae19e943 --- /dev/null +++ b/slatedb/src/wal/slatedb/gc.rs @@ -0,0 +1,400 @@ +use crate::db_state::SsTableId; +use crate::garbage_collector::stats::GcStats; +use crate::garbage_collector::{retain_allowed_by_gc_filter, GcFilter, GC_DELETE_CONCURRENCY}; +use crate::tablestore::TableStore; +use crate::wal::{WalError, WalFileRange, WalGc}; +use async_trait::async_trait; +use chrono::{DateTime, Utc}; +use futures::StreamExt; +use log::error; +use slatedb_common::clock::SystemClock; +use slatedb_common::object_metadata::IdentifiedObjectMetadata; +use std::ops::Bound; +use std::sync::Arc; +use std::time::Duration; + +/// Selects which class of SlateDB WAL object is collected. +/// +/// Regular WAL SSTs and zero-byte WAL fence objects share the same WAL +/// directory and `SsTableId::Wal` identifier space, but they have separate +/// retention policies. This mode keeps a single implementation while +/// allowing regular WAL GC and fence WAL GC to run on independent schedules. +#[derive(Debug, Clone, Copy)] +pub(crate) enum WalGcMode { + /// Collect non-empty WAL SSTs that are old enough for retention and unreferenced by active + /// manifests. + Regular, + + /// Collect zero-byte WAL fence objects under the same safety checks as regular WAL GC. + Fence, +} + +impl WalGcMode { + pub(crate) fn resource(self) -> &'static str { + match self { + WalGcMode::Regular => "WAL", + WalGcMode::Fence => "WAL fence", + } + } +} + +#[derive(Clone)] +pub(crate) struct SlateDbWalGc { + table_store: Arc, + stats: Arc, + mode: WalGcMode, + gc_filter: Option>, + system_clock: Arc, +} + +impl std::fmt::Debug for SlateDbWalGc { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("SlateDbWalGc") + .field("mode", &self.mode) + .finish() + } +} + +impl SlateDbWalGc { + pub(crate) fn new( + table_store: Arc, + stats: Arc, + mode: WalGcMode, + gc_filter: Option>, + system_clock: Arc, + ) -> Self { + Self { + table_store, + stats, + mode, + gc_filter, + system_clock, + } + } + + fn is_wal_sst_eligible_for_deletion( + utc_now: &DateTime, + wal_sst: &IdentifiedObjectMetadata, + min_age: &chrono::Duration, + referenced_ranges: &[WalFileRange], + ) -> bool { + if utc_now.signed_duration_since(wal_sst.metadata.last_modified) <= *min_age { + return false; + } + + let wal_sst_id = wal_sst.id.unwrap_wal_id(); + !referenced_ranges + .iter() + .any(|range| Self::range_contains(range, wal_sst_id)) + } + + fn range_contains(range: &WalFileRange, wal_sst_id: u64) -> bool { + let after_start = match &range.0 { + Bound::Included(start) => wal_sst_id >= *start, + Bound::Excluded(start) => wal_sst_id > *start, + Bound::Unbounded => true, + }; + let before_end = match &range.1 { + Bound::Included(end) => wal_sst_id <= *end, + Bound::Excluded(end) => wal_sst_id < *end, + Bound::Unbounded => true, + }; + after_start && before_end + } + + fn wal_sst_min_age(&self, min_age: Duration) -> chrono::Duration { + chrono::Duration::from_std(min_age).expect("invalid duration") + } + + /// Deletes the given WAL SSTs from the table store. + /// + /// In case of dryrun, the actual deletion doesn't happen. + async fn maybe_delete_wal_ssts(&self, sst_ids: Vec, dry_run: bool) { + if dry_run { + if !sst_ids.is_empty() { + log::info!( + "dry run: skipping {} deletion [count={}]", + self.mode.resource(), + sst_ids.len() + ); + if matches!(self.mode, WalGcMode::Fence) { + log::info!( + "WAL fence GC is dry-run by default. This is a conservative setting. \ + Set wal_fence_options.dry_run=false and use a conservative min_age to enable. \ + Silence this log with wal_fence_options=None. See #352 for details." + ); + } + } + for id in sst_ids { + log::debug!( + "dry run: would delete {} but skipped [id={:?}]", + self.mode.resource(), + id + ); + } + return; + } + + futures::stream::iter(sst_ids) + .for_each_concurrent(GC_DELETE_CONCURRENCY, |id| async move { + if let Err(e) = self.table_store.delete_sst(&id).await { + error!("error deleting WAL SST [id={:?}, error={}]", id, e); + } else { + match self.mode { + WalGcMode::Regular => self.stats.gc_wal_count.increment(1), + WalGcMode::Fence => self.stats.gc_wal_fence_count.increment(1), + } + } + }) + .await; + } +} + +#[async_trait] +impl WalGc for SlateDbWalGc { + async fn collect( + &self, + referenced_ranges: Vec, + min_age: Duration, + dry_run: bool, + ) -> Result<(), WalError> { + let utc_now = self.system_clock.now(); + let min_age = self.wal_sst_min_age(min_age); + let ssts_to_delete = self + .table_store + .list_wal_ssts(..) + .await? + .into_iter() + .filter(|wal_sst| match self.mode { + // In regular mode, only consider WAL SSTs with size > 0 for deletion. + WalGcMode::Regular => wal_sst.metadata.size > 0, + // In fence mode, only consider zero-byte WAL SSTs for deletion. + WalGcMode::Fence => wal_sst.metadata.size == 0, + }) + .filter(|wal_sst| { + Self::is_wal_sst_eligible_for_deletion( + &utc_now, + wal_sst, + &min_age, + &referenced_ranges, + ) + }) + .collect::>(); + let ssts_to_delete = retain_allowed_by_gc_filter(&self.gc_filter, ssts_to_delete).await; + let sst_ids_to_delete = ssts_to_delete + .into_iter() + .map(|wal_sst| wal_sst.id) + .collect::>(); + + self.maybe_delete_wal_ssts(sst_ids_to_delete, dry_run).await; + + Ok(()) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::block_cache_policy::BlockCachePolicy; + use crate::format::sst::SsTableFormat; + use crate::object_stores::ObjectStores; + use crate::tablestore::TableStoreKind; + use crate::RowEntry; + use object_store::memory::InMemory; + use object_store::path::Path; + use object_store::ObjectStore; + use slatedb_common::clock::MockSystemClock; + use slatedb_common::metrics::MetricsRecorderHelper; + use std::time::Duration; + + fn build_table_store() -> Arc { + let object_store: Arc = Arc::new(InMemory::new()); + Arc::new(TableStore::new( + ObjectStores::new(object_store, None), + SsTableFormat::default(), + Path::from("/"), + None, + TableStoreKind::GC, + BlockCachePolicy::default(), + )) + } + + fn build_collector( + table_store: Arc, + clock: Arc, + mode: WalGcMode, + ) -> SlateDbWalGc { + SlateDbWalGc::new( + table_store, + Arc::new(GcStats::new(&MetricsRecorderHelper::noop())), + mode, + None, + clock, + ) + } + + async fn write_regular_wal(table_store: &Arc, wal_id: u64) { + let mut sst = table_store.wal_table_builder(); + sst.add(RowEntry::new_value(b"key", b"value", wal_id)) + .await + .unwrap(); + let sst = sst.build().await.unwrap(); + table_store + .write_sst(&SsTableId::Wal(wal_id), &sst) + .await + .unwrap(); + } + + async fn write_fence_wal(table_store: &Arc, wal_id: u64) { + table_store.write_wal_fence(wal_id).await.unwrap(); + } + + async fn wal_ids(table_store: &Arc) -> Vec { + table_store + .list_wal_ssts(..) + .await + .unwrap() + .into_iter() + .map(|wal| wal.id.unwrap_wal_id()) + .collect() + } + + async fn make_all_wals_older_than( + table_store: &Arc, + clock: &MockSystemClock, + min_age: Duration, + ) { + let newest_wal = table_store + .list_wal_ssts(..) + .await + .unwrap() + .into_iter() + .map(|wal| wal.metadata.last_modified) + .max() + .expect("expected at least one WAL"); + let min_age_millis = + i64::try_from(min_age.as_millis()).expect("min_age should fit in i64 milliseconds"); + clock.set(newest_wal.timestamp_millis() + min_age_millis + 1_000); + } + + fn protect_outer_wals() -> Vec { + vec![ + WalFileRange(Bound::Included(1), Bound::Excluded(2)), + WalFileRange(Bound::Included(4), Bound::Unbounded), + ] + } + + #[tokio::test] + async fn regular_mode_deletes_unreferenced_range_and_keeps_referenced_wals() { + let table_store = build_table_store(); + let clock = Arc::new(MockSystemClock::new()); + for wal_id in 1..=4 { + write_regular_wal(&table_store, wal_id).await; + } + make_all_wals_older_than(&table_store, &clock, Duration::ZERO).await; + let collector = build_collector(table_store.clone(), clock, WalGcMode::Regular); + + collector + .collect(protect_outer_wals(), Duration::ZERO, false) + .await + .unwrap(); + + assert_eq!(wal_ids(&table_store).await, vec![1, 4]); + } + + #[tokio::test] + async fn regular_mode_does_not_touch_fence_wals() { + let table_store = build_table_store(); + let clock = Arc::new(MockSystemClock::new()); + write_regular_wal(&table_store, 1).await; + write_fence_wal(&table_store, 2).await; + make_all_wals_older_than(&table_store, &clock, Duration::ZERO).await; + let collector = build_collector(table_store.clone(), clock, WalGcMode::Regular); + + collector + .collect(vec![], Duration::ZERO, false) + .await + .unwrap(); + + assert_eq!(wal_ids(&table_store).await, vec![2]); + } + + #[tokio::test] + async fn regular_mode_respects_min_age() { + let table_store = build_table_store(); + let clock = Arc::new(MockSystemClock::new()); + write_regular_wal(&table_store, 1).await; + let last_modified = table_store + .metadata(&SsTableId::Wal(1)) + .await + .unwrap() + .last_modified; + let min_age = Duration::from_secs(60 * 60); + let collector = build_collector(table_store.clone(), clock.clone(), WalGcMode::Regular); + + clock.set((last_modified + chrono::Duration::minutes(30)).timestamp_millis()); + collector.collect(vec![], min_age, false).await.unwrap(); + assert_eq!(wal_ids(&table_store).await, vec![1]); + + clock.set((last_modified + chrono::Duration::minutes(61)).timestamp_millis()); + collector.collect(vec![], min_age, false).await.unwrap(); + assert!(wal_ids(&table_store).await.is_empty()); + } + + #[tokio::test] + async fn fence_mode_deletes_unreferenced_range_and_keeps_referenced_wals() { + let table_store = build_table_store(); + let clock = Arc::new(MockSystemClock::new()); + for wal_id in 1..=4 { + write_fence_wal(&table_store, wal_id).await; + } + make_all_wals_older_than(&table_store, &clock, Duration::ZERO).await; + let collector = build_collector(table_store.clone(), clock, WalGcMode::Fence); + + collector + .collect(protect_outer_wals(), Duration::ZERO, false) + .await + .unwrap(); + + assert_eq!(wal_ids(&table_store).await, vec![1, 4]); + } + + #[tokio::test] + async fn fence_mode_does_not_touch_regular_wals() { + let table_store = build_table_store(); + let clock = Arc::new(MockSystemClock::new()); + write_fence_wal(&table_store, 1).await; + write_regular_wal(&table_store, 2).await; + make_all_wals_older_than(&table_store, &clock, Duration::ZERO).await; + let collector = build_collector(table_store.clone(), clock, WalGcMode::Fence); + + collector + .collect(vec![], Duration::ZERO, false) + .await + .unwrap(); + + assert_eq!(wal_ids(&table_store).await, vec![2]); + } + + #[tokio::test] + async fn fence_mode_respects_min_age() { + let table_store = build_table_store(); + let clock = Arc::new(MockSystemClock::new()); + write_fence_wal(&table_store, 1).await; + let last_modified = table_store + .metadata(&SsTableId::Wal(1)) + .await + .unwrap() + .last_modified; + let min_age = Duration::from_secs(60 * 60); + let collector = build_collector(table_store.clone(), clock.clone(), WalGcMode::Fence); + + clock.set((last_modified + chrono::Duration::minutes(30)).timestamp_millis()); + collector.collect(vec![], min_age, false).await.unwrap(); + assert_eq!(wal_ids(&table_store).await, vec![1]); + + clock.set((last_modified + chrono::Duration::minutes(61)).timestamp_millis()); + collector.collect(vec![], min_age, false).await.unwrap(); + assert!(wal_ids(&table_store).await.is_empty()); + } +} diff --git a/slatedb/src/wal/slatedb/iterator.rs b/slatedb/src/wal/slatedb/iterator.rs new file mode 100644 index 0000000000..289cecec68 --- /dev/null +++ b/slatedb/src/wal/slatedb/iterator.rs @@ -0,0 +1,731 @@ +use std::collections::VecDeque; +use std::sync::Arc; +use std::time::Duration; + +use async_trait::async_trait; +use log::error; +use slatedb_common::clock::SystemClock; +use tokio::sync::watch; +use tokio::task; +use tokio::task::JoinHandle; + +use crate::db_status::DbStatus; +use crate::error::SlateDBError; +use crate::iter::{EmptyIterator, RowEntryIterator}; +use crate::manifest::store::ManifestStore; +use crate::manifest::VersionedManifest; +use crate::utils::panic_string; +use crate::wal::{WalError, WalIterator as WalIteratorTrait, WalRows}; +use crate::RowEntry; + +use super::sst_iterator::{WalSstIterator, WalSstIteratorOptions}; +use super::store::WalTableStore; + +#[async_trait] +pub(crate) trait ManifestReader: Send + Sync + 'static { + async fn manifest(&self) -> Result; +} + +#[async_trait] +impl ManifestReader for watch::Receiver { + async fn manifest(&self) -> Result { + Ok(self.borrow().current_manifest.clone()) + } +} + +#[async_trait] +impl ManifestReader for ManifestStore { + async fn manifest(&self) -> Result { + self.read_latest_manifest().await + } +} + +pub(crate) struct SlateDbWalIteratorOptions { + /// The number of WAL SSTs to preload while replaying. + pub(crate) sst_batch_size: usize, + + /// Options to pass through to the underlying WAL SST iterators. + pub(crate) sst_iter_options: WalSstIteratorOptions, +} + +impl Default for SlateDbWalIteratorOptions { + fn default() -> Self { + Self { + sst_batch_size: 4, + sst_iter_options: WalSstIteratorOptions::default(), + } + } +} + +enum WalFileIterator { + Empty(EmptyIterator), + Sst(Box), +} + +impl WalFileIterator { + async fn next(&mut self) -> Result, SlateDBError> { + match self { + Self::Empty(iter) => iter.next().await, + Self::Sst(iter) => iter.next().await, + } + } +} + +struct WalRowsCollector { + wal_id: u64, + iter: WalFileIterator, + rows: Vec, + drained: bool, +} + +impl WalRowsCollector { + fn new(wal_id: u64, iter: WalFileIterator) -> Self { + Self { + wal_id, + iter, + rows: vec![], + drained: false, + } + } + + async fn collect(&mut self) -> Result<(), WalError> { + loop { + match self.iter.next().await { + Ok(Some(row)) => self.rows.push(row), + Ok(None) => { + self.drained = true; + break Ok(()); + } + Err(err) if err.has_object_store_not_found() => { + break Err(WalError::WalTruncated(self.wal_id)); + } + Err(err) => { + break Err(err.into()); + } + } + } + } +} + +impl From for WalRows { + fn from(reader: WalRowsCollector) -> Self { + assert!(reader.drained); + WalRows { + last_consumed_wal_file_id: reader.wal_id, + rows: reader.rows, + } + } +} + +struct CurrentWalFile { + initialized: bool, + collector: Option, +} + +impl CurrentWalFile { + fn initial() -> Self { + Self { + initialized: false, + collector: None, + } + } + + fn initialized(&self) -> bool { + self.initialized + } + + async fn collect(&mut self) -> Result, WalError> { + assert!(self.initialized); + let Some(collector) = &mut self.collector else { + return Ok(None); + }; + collector.collect().await?; + let collector = self.collector.take().expect("unreachable"); + self.initialized = false; + Ok(Some(collector.into())) + } + + fn advance(&mut self, collector: WalRowsCollector) { + assert!(!self.initialized); + self.initialized = true; + self.collector = Some(collector); + } + + fn finish(&mut self) { + self.initialized = true; + self.collector = None; + } +} + +/// Iterates over the writes in a range of WAL files, preloading up to +/// `sst_batch_size` WAL SST handles concurrently. Returns the rows of one WAL +/// file per [`WalRows`], and verifies that files carry strictly increasing seq +/// ranges — the ordering callers rely on to split and tag memtables safely. +/// +/// A file's rows are read sequentially only when it is returned from +/// [`Self::next`], so at most one file's rows are materialized at a time. For an +/// unbounded end, open tasks poll their assigned future WAL IDs until the files +/// appear or the manifest proves that a missing file was truncated. +pub(crate) struct SlateDbWalIterator { + options: SlateDbWalIteratorOptions, + end_bound: WalIteratorEndBound, + wal_store: Arc, + next_files: VecDeque>>, + next_wal_id: Option, + /// The greatest seq returned so far, used to verify that WAL files arrive + /// with strictly increasing seq ranges. + last_seq: Option, + /// Set once iteration has ended, either because the range was exhausted or + /// because an error was returned. + terminal_result: Option, WalError>>, + current_file: CurrentWalFile, +} + +#[derive(Clone)] +pub(crate) enum WalIteratorEndBound { + Exclusive(u64), + Unbounded { + manifest_reader: Arc, + poll_interval: Duration, + system_clock: Arc, + }, +} + +impl WalIteratorEndBound { + fn contains(&self, wal_id: u64) -> bool { + match self { + Self::Exclusive(end_wal_id) => wal_id < *end_wal_id, + Self::Unbounded { .. } => true, + } + } +} + +impl std::fmt::Debug for WalIteratorEndBound { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self { + Self::Exclusive(end_wal_id) => f.debug_tuple("Exclusive").field(end_wal_id).finish(), + Self::Unbounded { poll_interval, .. } => f + .debug_struct("Unbounded") + .field("poll_interval", poll_interval) + .finish_non_exhaustive(), + } + } +} + +impl SlateDbWalIterator { + pub(crate) fn range( + from_wal_id: u64, + to_bound: WalIteratorEndBound, + options: SlateDbWalIteratorOptions, + wal_store: Arc, + ) -> Result { + if options.sst_batch_size < 1 { + return Err(SlateDBError::InvalidSSTBatchSize(options.sst_batch_size)); + } + + Ok(Self { + options, + end_bound: to_bound, + wal_store, + next_files: VecDeque::new(), + next_wal_id: Some(from_wal_id), + last_seq: None, + terminal_result: None, + current_file: CurrentWalFile::initial(), + }) + } + + fn spawn_opens(&mut self) { + while self.maybe_spawn_open() {} + } + + fn maybe_spawn_open(&mut self) -> bool { + let Some(next_wal_id) = self.next_wal_id else { + return false; + }; + if !self.end_bound.contains(next_wal_id) + || self.next_files.len() >= self.options.sst_batch_size + { + return false; + } + + self.next_wal_id = next_wal_id.checked_add(1); + + async fn try_open_file_iter( + wal_id: u64, + sst_iter_options: WalSstIteratorOptions, + wal_store: Arc, + ) -> Result { + let sst = match wal_store.open_sst(wal_id).await { + Ok(sst) => sst, + Err(SlateDBError::EmptySSTable) => { + // Zero-byte WAL files are fence markers; replay them as empty WALs + // so the last replayed WAL ID still advances past the marker. + return Ok(WalRowsCollector::new( + wal_id, + WalFileIterator::Empty(EmptyIterator::new()), + )); + } + Err(err) => return Err(err), + }; + let iter = WalSstIterator::new(sst, Arc::clone(&wal_store), sst_iter_options).await?; + Ok(WalRowsCollector::new( + wal_id, + WalFileIterator::Sst(Box::new(iter)), + )) + } + + async fn open_file_iter( + wal_id: u64, + sst_iter_options: WalSstIteratorOptions, + wal_store: Arc, + end_bound: WalIteratorEndBound, + ) -> Result { + loop { + match try_open_file_iter(wal_id, sst_iter_options.clone(), Arc::clone(&wal_store)) + .await + { + Ok(iter) => return Ok(iter), + Err(err) if err.has_object_store_not_found() => { + let WalIteratorEndBound::Unbounded { + manifest_reader, + poll_interval, + system_clock, + } = &end_bound + else { + return Err(WalError::WalTruncated(wal_id)); + }; + + let manifest = manifest_reader.manifest().await?; + if wal_id < manifest.next_wal_sst_id() { + // This WAL is known to have been written durably in the past, + // so it must have been deleted by GC. + return Err(WalError::WalTruncated(wal_id)); + } + system_clock.sleep(*poll_interval).await; + } + Err(err) => return Err(err.into()), + } + } + } + + let handle = task::spawn(open_file_iter( + next_wal_id, + self.options.sst_iter_options.clone(), + Arc::clone(&self.wal_store), + self.end_bound.clone(), + )); + self.next_files.push_back(handle); + true + } + + fn open_task_result( + end_bound: &WalIteratorEndBound, + result: Result, task::JoinError>, + ) -> Result { + match result { + Ok(result) => result, + Err(join_err) => { + let task_name = format!("wal_replay[end_bound={end_bound:?}]"); + let msg = if let Ok(panic_err) = join_err.try_into_panic() { + format!( + "wal_replay task panicked unexpectedly. [task_name={}, panic={}]", + task_name, + panic_string(&panic_err), + ) + } else { + format!("wal_replay task cancelled. [task_name={}]", task_name) + }; + error!("{}", msg); + let error = Arc::from(Box::::from(msg)); + Err(WalError::InternalError(error)) + } + } + } + + /// Opens WAL handles in the background and promotes the oldest file when a + /// current file is needed. + async fn load_next_file(&mut self) -> Result<(), WalError> { + if self.current_file.initialized() { + return Ok(()); + } + + // Populate the pre-load queue first to handle the case where it's initially empty + self.spawn_opens(); + // await a mutable ref to the task so that next remains cancel-safe + // see https://docs.rs/tokio/latest/tokio/task/struct.JoinHandle.html#cancel-safety + let Some(join_handle) = self.next_files.front_mut() else { + self.current_file.finish(); + return Ok(()); + }; + let result = join_handle.await; + self.next_files.pop_front(); + // Refill the preload queue before returning so the iterator starts loading the next file + self.spawn_opens(); + self.current_file + .advance(Self::open_task_result(&self.end_bound, result)?); + Ok(()) + } + + fn terminate( + &mut self, + result: Result, WalError>, + ) -> Result, WalError> { + self.terminal_result = Some(result.clone()); + for task in self.next_files.drain(..) { + task.abort(); + } + self.current_file.collector = None; + result + } +} + +#[async_trait] +impl WalIteratorTrait for SlateDbWalIterator { + /// Get the next set of writes from the WAL files in the range. Each returned + /// [`WalRows`] holds the rows of one WAL file; a WAL file with no rows + /// yields a batch with empty `rows`. A bounded iterator returns `None` once + /// its range has been read; an iterator with an unbounded end polls future + /// WAL files instead. + async fn next(&mut self) -> Result, WalError> { + if let Some(result) = self.terminal_result.clone() { + return result; + } + + if let Err(err) = self.load_next_file().await { + return self.terminate(Err(err)); + } + match self.current_file.collect().await { + Err(err) => self.terminate(Err(err)), + Ok(None) => self.terminate(Ok(None)), + Ok(Some(rows)) => { + // Verify that WAL files carry strictly increasing seq ranges. Replay + // relies on this ordering to split and tag memtables safely: a commit seq + // spanning two WAL files, or files with overlapping seq ranges, would + // break recovery's (wal_id, seq) watermark filtering. + if let Some(min_seq) = rows.rows.iter().map(|row| row.seq).min() { + if let Some(last_seq) = self.last_seq { + if min_seq <= last_seq { + let msg = format!( + "WAL replay saw out-of-order seqs across WAL files. \ + [wal_id={}, min_seq={}, last_seq={}]", + rows.last_consumed_wal_file_id, min_seq, last_seq, + ); + error!("{}", msg); + let error = + Arc::from(Box::::from(msg)); + return self.terminate(Err(WalError::InternalError(error))); + } + } + let max_seq = rows + .rows + .iter() + .map(|row| row.seq) + .max() + .expect("non-empty rows have a max seq"); + self.last_seq = Some(max_seq); + } + Ok(Some(rows)) + } + } + } +} + +impl Drop for SlateDbWalIterator { + fn drop(&mut self) { + for task in self.next_files.drain(..) { + task.abort(); + } + } +} + +#[cfg(test)] +mod tests { + use std::collections::{BTreeMap, BTreeSet}; + use std::sync::Arc; + use std::time::Duration; + + use bytes::Bytes; + use object_store::memory::InMemory; + use object_store::path::Path; + use object_store::ObjectStore; + use slatedb_common::clock::DefaultSystemClock; + + use super::{SlateDbWalIterator, SlateDbWalIteratorOptions, WalIteratorEndBound}; + use crate::db_status::DbStatusManager; + use crate::format::sst::SsTableFormat; + use crate::manifest::{Manifest, ManifestCore, VersionedManifest}; + use crate::object_store_tag::TableStoreKind; + use crate::types::RowEntry; + use crate::wal::slatedb::store::WalTableStore; + use crate::wal::{WalError, WalIterator as _}; + + fn versioned_manifest(id: u64, next_wal_id: u64) -> VersionedManifest { + let mut core = ManifestCore::new(); + core.next_wal_sst_id = next_wal_id; + VersionedManifest::from_manifest(id, Manifest::initial(core)) + } + + fn status_manager(next_wal_id: u64) -> DbStatusManager { + DbStatusManager::new_with_initial_values( + 0, + versioned_manifest(1, next_wal_id), + BTreeSet::new(), + ) + } + + #[tokio::test] + async fn should_repeat_terminal_error_for_wal_iterator() { + let table_store = test_table_store(); + let mut wal_iter = SlateDbWalIterator::range( + 1, + WalIteratorEndBound::Exclusive(2), + SlateDbWalIteratorOptions::default(), + Arc::clone(&table_store), + ) + .unwrap(); + + assert!(matches!( + wal_iter.next().await, + Err(WalError::WalTruncated(1)) + )); + assert!(matches!( + wal_iter.next().await, + Err(WalError::WalTruncated(1)) + )); + } + + #[tokio::test] + async fn should_repeat_terminal_none_for_wal_iterator() { + let table_store = test_table_store(); + let mut wal_iter = SlateDbWalIterator::range( + 1, + WalIteratorEndBound::Exclusive(1), + SlateDbWalIteratorOptions::default(), + Arc::clone(&table_store), + ) + .unwrap(); + + assert!(wal_iter.next().await.unwrap().is_none()); + assert!(wal_iter.next().await.unwrap().is_none()); + } + + #[tokio::test] + async fn should_honor_an_exclusive_end_bound() { + let table_store = test_table_store(); + table_store.write_wal_fence(1).await.unwrap(); + table_store.write_wal_fence(2).await.unwrap(); + let mut wal_iter = SlateDbWalIterator::range( + 1, + WalIteratorEndBound::Exclusive(2), + SlateDbWalIteratorOptions::default(), + table_store, + ) + .unwrap(); + + let batch = wal_iter.next().await.unwrap().unwrap(); + assert_eq!(batch.last_consumed_wal_file_id, 1); + assert!(batch.rows.is_empty()); + assert!(wal_iter.next().await.unwrap().is_none()); + } + + #[tokio::test(start_paused = true)] + async fn should_poll_future_wals_in_an_unbounded_range() { + let table_store = test_table_store(); + let status_manager = status_manager(1); + let mut wal_iter = SlateDbWalIterator::range( + 1, + WalIteratorEndBound::Unbounded { + manifest_reader: Arc::new(status_manager.subscribe()), + poll_interval: Duration::from_millis(10), + system_clock: Arc::new(DefaultSystemClock::new()), + }, + SlateDbWalIteratorOptions::default(), + Arc::clone(&table_store), + ) + .unwrap(); + + assert!( + tokio::time::timeout(Duration::from_millis(30), wal_iter.next()) + .await + .is_err(), + "an unbounded iterator returned before WAL 1 existed" + ); + + table_store.write_wal_fence(1).await.unwrap(); + let first = tokio::time::timeout(Duration::from_millis(100), wal_iter.next()) + .await + .expect("iterator did not observe WAL 1") + .unwrap() + .expect("unbounded iterator returned None"); + assert!(first.rows.is_empty()); + assert_eq!(first.last_consumed_wal_file_id, 1); + + assert!( + tokio::time::timeout(Duration::from_millis(30), wal_iter.next()) + .await + .is_err(), + "an unbounded iterator returned before WAL 2 existed" + ); + + table_store.write_wal_fence(2).await.unwrap(); + let second = tokio::time::timeout(Duration::from_millis(100), wal_iter.next()) + .await + .expect("iterator did not observe WAL 2") + .unwrap() + .expect("unbounded iterator returned None"); + assert!(second.rows.is_empty()); + assert_eq!(second.last_consumed_wal_file_id, 2); + } + + #[tokio::test(start_paused = true)] + async fn should_report_truncation_when_manifest_advances_past_a_missing_wal() { + let table_store = test_table_store(); + let status_manager = status_manager(1); + let mut wal_iter = SlateDbWalIterator::range( + 1, + WalIteratorEndBound::Unbounded { + manifest_reader: Arc::new(status_manager.subscribe()), + poll_interval: Duration::from_millis(10), + system_clock: Arc::new(DefaultSystemClock::new()), + }, + SlateDbWalIteratorOptions::default(), + table_store, + ) + .unwrap(); + + assert!( + tokio::time::timeout(Duration::from_millis(30), wal_iter.next()) + .await + .is_err(), + "the iterator did not poll a future WAL" + ); + + status_manager.report_manifest(versioned_manifest(2, 2)); + let result = tokio::time::timeout(Duration::from_millis(100), wal_iter.next()) + .await + .expect("iterator did not react to the manifest update"); + assert!(matches!(result, Err(WalError::WalTruncated(1)))); + } + + #[tokio::test] + async fn should_return_atomic_wal_rows_in_increasing_seq_order() { + let table_store = test_table_store(); + // Each file contains out-of-order rows and a sequence that appears twice. + // The iterator must keep both rows for a sequence in one batch, while the + // sequence range of the second batch must follow the first. + let wal_entries = [ + vec![ + RowEntry::new_value(b"key_001", &[b'x'; 128], 2), + RowEntry::new_value(b"key_002", &[b'x'; 128], 1), + RowEntry::new_value(b"key_003", &[b'x'; 128], 2), + ], + vec![ + RowEntry::new_value(b"key_004", &[b'x'; 128], 4), + RowEntry::new_value(b"key_005", &[b'x'; 128], 3), + RowEntry::new_value(b"key_006", &[b'x'; 128], 4), + ], + ]; + let mut expected_rows = BTreeMap::new(); + let mut expected_rows_by_seq = BTreeMap::>::new(); + let wal_file_count = wal_entries.len() as u64; + for (file_index, entries) in wal_entries.iter().enumerate() { + let wal_id = file_index as u64 + 1; + for row in entries { + expected_rows.insert(row.key.clone(), (row.clone(), wal_id)); + expected_rows_by_seq + .entry(row.seq) + .or_default() + .insert(row.key.clone()); + } + } + for (index, entries) in wal_entries.into_iter().enumerate() { + let mut builder = table_store.table_builder(); + for entry in entries { + builder.add(entry).await.unwrap(); + } + let encoded_sst = builder.build().await.unwrap(); + table_store + .write_sst(index as u64 + 1, &encoded_sst) + .await + .unwrap(); + } + let mut wal_iter = SlateDbWalIterator::range( + 1, + WalIteratorEndBound::Exclusive(wal_file_count + 1), + SlateDbWalIteratorOptions::default(), + Arc::clone(&table_store), + ) + .unwrap(); + + let mut returned_rows = BTreeMap::new(); + let mut previous_max_seq = None; + let mut last_consumed_wal_file_id = 0; + while let Some(batch) = wal_iter.next().await.unwrap() { + let batch_min_seq = batch.rows.iter().map(|r| r.seq).min().unwrap(); + let batch_max_seq = batch.rows.iter().map(|r| r.seq).max().unwrap(); + if let Some(previous_max_seq) = previous_max_seq { + assert!( + batch_min_seq > previous_max_seq, + "consecutive WAL batches have overlapping sequence ranges" + ); + } + previous_max_seq = Some(batch_max_seq); + + let mut batch_rows_by_seq = BTreeMap::>::new(); + for row in &batch.rows { + assert!( + returned_rows.insert(row.key.clone(), row.clone()).is_none(), + "row was returned more than once: {:?}", + row.key + ); + batch_rows_by_seq + .entry(row.seq) + .or_default() + .insert(row.key.clone()); + } + for (seq, batch_rows) in batch_rows_by_seq { + assert_eq!( + expected_rows_by_seq.get(&seq), + Some(&batch_rows), + "rows for seq {seq} were split across WAL batches" + ); + } + + assert!( + batch.last_consumed_wal_file_id >= last_consumed_wal_file_id, + "consumed WAL file watermark moved backwards" + ); + assert!(batch.last_consumed_wal_file_id <= wal_file_count); + for wal_id in 1..=batch.last_consumed_wal_file_id { + let file_fully_returned = + expected_rows.iter().all(|(key, (_, expected_wal_id))| { + *expected_wal_id != wal_id || returned_rows.contains_key(key) + }); + assert!( + file_fully_returned, + "WAL file {wal_id} was marked consumed before all its rows were returned" + ); + } + last_consumed_wal_file_id = batch.last_consumed_wal_file_id; + } + + let expected_returned_rows = expected_rows + .into_iter() + .map(|(key, (row, _wal_id))| (key, row)) + .collect(); + assert_eq!(returned_rows, expected_returned_rows); + assert_eq!(last_consumed_wal_file_id, wal_file_count); + } + + fn test_table_store() -> Arc { + let object_store: Arc = Arc::new(InMemory::new()); + let path = Path::from("/tmp/test_kv_store"); + Arc::new(WalTableStore::new( + object_store, + SsTableFormat::default(), + path, + TableStoreKind::Main, + )) + } +} diff --git a/slatedb/src/wal/slatedb/mod.rs b/slatedb/src/wal/slatedb/mod.rs new file mode 100644 index 0000000000..2c42583a4c --- /dev/null +++ b/slatedb/src/wal/slatedb/mod.rs @@ -0,0 +1,11 @@ +//! SlateDB's native object-store-backed WAL implementation. + +pub(crate) mod admin; +pub(crate) mod gc; +pub(crate) mod iterator; +pub(crate) mod reader; +pub(crate) mod sst_builder; +pub(crate) mod sst_iterator; +pub(crate) mod store; +pub(crate) mod writer; +pub(crate) mod writer_init; diff --git a/slatedb/src/wal/slatedb/reader.rs b/slatedb/src/wal/slatedb/reader.rs new file mode 100644 index 0000000000..b17d5c787f --- /dev/null +++ b/slatedb/src/wal/slatedb/reader.rs @@ -0,0 +1,706 @@ +use std::ops::Bound; +use std::sync::Arc; +use std::time::Duration; + +use async_trait::async_trait; +use log::error; +use object_store::{path::Path, ObjectStore}; +use slatedb_common::clock::{DefaultSystemClock, SystemClock}; + +use crate::db_status::DbStatusManager; +use crate::error::SlateDBError; +use crate::format::sst::SsTableFormat; +use crate::manifest::store::ManifestStore; +use crate::object_store_tag::TableStoreKind; +use crate::wal::slatedb::iterator::{ + ManifestReader, SlateDbWalIterator, SlateDbWalIteratorOptions, WalIteratorEndBound, +}; +use crate::wal::slatedb::store::WalTableStore; +use crate::wal::{WalError, WalFileRange, WalIterator, WalReader}; + +#[derive(Clone, Debug)] +pub struct SlateDbWalReaderOptions { + /// The number of WAL SST handles to preload while replaying. + pub sst_batch_size: usize, + + /// Retained for compatibility with the existing WAL reader configuration. + pub max_fetch_tasks: usize, + + /// The target number of bytes to fetch in a single request while iterating over WAL SSTs. + /// Each fetch reads enough whole blocks to meet this target or reach the end of the file. + /// The default is 1 MiB. + pub read_ahead_bytes: usize, +} + +impl Default for SlateDbWalReaderOptions { + fn default() -> Self { + Self { + sst_batch_size: 4, + max_fetch_tasks: 2, + read_ahead_bytes: 64 * 1024 * 1024, + } + } +} + +impl From for SlateDbWalIteratorOptions { + fn from(options: SlateDbWalReaderOptions) -> Self { + Self { + sst_batch_size: options.sst_batch_size, + sst_iter_options: super::sst_iterator::WalSstIteratorOptions { + target_bytes_to_fetch: options.read_ahead_bytes, + }, + } + } +} + +/// Builder for a [`SlateDbWalReader`]. +/// +/// Callers must configure both the database path with [`Self::with_path`] and +/// the primary object store with [`Self::with_object_store`]. By default, the +/// primary object store is used for both the manifest and WAL files. Use +/// [`Self::with_wal_object_store`] when WAL files are stored separately. +/// +/// Reader options and the system clock use their defaults when they are not +/// explicitly configured. +pub struct SlateDbWalReaderBuilder { + path: Option, + wal_store: Option>, + object_store: Option>, + wal_object_store: Option>, + manifest_reader: Option>, + system_clock: Arc, + options: SlateDbWalReaderOptions, +} + +impl Default for SlateDbWalReaderBuilder { + fn default() -> Self { + Self { + path: None, + wal_store: None, + object_store: None, + wal_object_store: None, + manifest_reader: None, + system_clock: Arc::new(DefaultSystemClock::new()), + options: SlateDbWalReaderOptions::default(), + } + } +} + +impl SlateDbWalReaderBuilder { + /// Creates a builder with default reader options and system clock. + pub fn new() -> Self { + Self::default() + } + + /// Sets the root path of the database to read. + pub fn with_path(mut self, path: Path) -> Self { + self.path = Some(path); + self + } + + /// Sets an existing WAL table store for internal construction. + pub(crate) fn with_wal_store(mut self, wal_store: Arc) -> Self { + self.wal_store = Some(wal_store); + self + } + + /// Sets the primary object store used to read the manifest and, by + /// default, WAL files. + pub fn with_object_store(mut self, object_store: Arc) -> Self { + self.object_store = Some(object_store); + self + } + + /// Sets a dedicated object store from which WAL files are read. + /// + /// The primary object store configured by [`Self::with_object_store`] + /// remains the source for the database manifest. + pub fn with_wal_object_store(mut self, wal_object_store: Arc) -> Self { + self.wal_object_store = Some(wal_object_store); + self + } + + /// Sets the clock used to wait between polls by live WAL iterators. + pub fn with_system_clock(mut self, system_clock: Arc) -> Self { + self.system_clock = system_clock; + self + } + + /// Sets the options controlling how WAL files are read. + pub fn with_options(mut self, options: SlateDbWalReaderOptions) -> Self { + self.options = options; + self + } + + /// Sets an existing manifest reader for internal construction. + pub(crate) fn with_manifest_reader(mut self, manifest_reader: Arc) -> Self { + self.manifest_reader = Some(manifest_reader); + self + } + + /// Builds a WAL reader from the configured state. + /// + /// # Errors + /// + /// Returns an invalid-configuration error when the database path or + /// primary object store has not been configured. Internal callers may + /// instead provide both a WAL store and manifest reader. + pub fn build(self) -> Result { + let manifest_reader = match self.manifest_reader { + Some(manifest_reader) => manifest_reader, + None => { + let Some(object_store) = self.object_store.clone() else { + return Err(crate::Error::invalid( + "must specify object store".to_string(), + )); + }; + let Some(path) = self.path.clone() else { + return Err(crate::Error::invalid("must specify db path".to_string())); + }; + Arc::new(ManifestStore::new(&path, object_store)) + } + }; + let wal_store = match self.wal_store { + Some(wal_store) => wal_store, + None => { + let Some(object_store) = self.object_store.clone() else { + return Err(crate::Error::invalid( + "must specify object store".to_string(), + )); + }; + let Some(path) = self.path.clone() else { + return Err(crate::Error::invalid("must specify db path".to_string())); + }; + let object_store = self.wal_object_store.unwrap_or(object_store); + Arc::new(WalTableStore::new( + object_store, + SsTableFormat::default(), + path.clone(), + TableStoreKind::Reader, + )) + } + }; + Ok(SlateDbWalReader { + wal_store, + manifest_reader, + system_clock: self.system_clock, + options: self.options, + }) + } +} + +pub struct SlateDbWalReader { + wal_store: Arc, + manifest_reader: Arc, + system_clock: Arc, + options: SlateDbWalReaderOptions, +} + +impl SlateDbWalReader { + pub(crate) fn new_with_status_manager( + wal_store: Arc, + db_status: &DbStatusManager, + system_clock: Arc, + options: SlateDbWalReaderOptions, + ) -> Self { + let manifest_reader: Arc = Arc::new(db_status.subscribe()); + SlateDbWalReaderBuilder::new() + .with_wal_store(wal_store) + .with_manifest_reader(manifest_reader) + .with_system_clock(system_clock) + .with_options(options) + .build() + .expect("WAL store and manifest reader initialize a WAL reader") + } +} + +#[async_trait] +impl WalReader for SlateDbWalReader { + async fn iterator( + &self, + wal_file_id_range: WalFileRange, + ) -> Result, WalError> { + let from_wal_id = match wal_file_id_range.0 { + Bound::Included(wal_id) => wal_id, + Bound::Excluded(wal_id) => wal_id.checked_add(1).ok_or_else(|| { + error!( + "WAL iterator start bound overflowed. [range={:?}]", + wal_file_id_range + ); + SlateDBError::InvalidDBState + })?, + Bound::Unbounded => { + error!( + "WAL iterator range must have a bounded start. [range={:?}]", + wal_file_id_range + ); + return Err(SlateDBError::InvalidDBState.into()); + } + }; + let end_bound = match wal_file_id_range.1 { + Bound::Included(wal_id) => { + WalIteratorEndBound::Exclusive(wal_id.checked_add(1).ok_or_else(|| { + error!( + "WAL iterator end bound overflowed. [range={:?}]", + wal_file_id_range + ); + SlateDBError::InvalidDBState + })?) + } + Bound::Excluded(wal_id) => WalIteratorEndBound::Exclusive(wal_id), + Bound::Unbounded => WalIteratorEndBound::Unbounded { + manifest_reader: Arc::clone(&self.manifest_reader), + poll_interval: Duration::from_secs(1), + system_clock: Arc::clone(&self.system_clock), + }, + }; + let iterator = SlateDbWalIterator::range( + from_wal_id, + end_bound, + self.options.clone().into(), + Arc::clone(&self.wal_store), + )?; + Ok(Box::new(iterator)) + } + + async fn last_wal_file_id(&self, replay_after_wal_id: u64) -> Result { + let last = self.wal_store.last_seen_wal_id(replay_after_wal_id).await?; + let manifest = self.manifest_reader.manifest().await?; + if last < manifest.core().replay_after_wal_id { + return Err(WalError::WalTruncated(last)); + } + Ok(last) + } +} + +#[cfg(test)] +mod tests { + use std::ops::Bound; + use std::time::Duration; + + use super::*; + use crate::config::{FlushOptions, FlushType}; + use crate::db_state::SsTableId; + use crate::manifest::store::StoredManifest; + use crate::manifest::ManifestCore; + use crate::paths::PathResolver; + use crate::test_utils::StringConcatMergeOperator; + use crate::types::ValueDeletable; + use crate::wal::WalRows; + use crate::Db; + use object_store::memory::InMemory; + use object_store::ObjectStoreExt; + fn end_after(wal_id: u64) -> u64 { + wal_id.checked_add(1).expect("test WAL ID overflow") + } + + fn assert_invalid_build(builder: SlateDbWalReaderBuilder, expected_message: &str) { + let error = match builder.build() { + Ok(_) => panic!("expected WAL reader builder to reject incomplete state"), + Err(error) => error, + }; + assert_eq!(error.kind(), crate::ErrorKind::Invalid); + assert!( + error.to_string().contains(expected_message), + "expected error containing {expected_message:?}, got {error}" + ); + } + + #[test] + fn builder_rejects_missing_object_store() { + assert_invalid_build( + SlateDbWalReaderBuilder::new().with_path(Path::from("/missing-object-store")), + "must specify object store", + ); + } + + #[test] + fn builder_rejects_missing_db_path() { + assert_invalid_build( + SlateDbWalReaderBuilder::new().with_object_store(Arc::new(InMemory::new())), + "must specify db path", + ); + } + + #[test] + fn builder_rejects_wal_store_without_manifest_reader() { + let object_store: Arc = Arc::new(InMemory::new()); + let wal_store = Arc::new(WalTableStore::new( + object_store, + SsTableFormat::default(), + Path::from("/table-store-without-manifest-reader"), + TableStoreKind::Reader, + )); + + assert_invalid_build( + SlateDbWalReaderBuilder::new().with_wal_store(wal_store), + "must specify object store", + ); + } + + #[test] + fn builder_rejects_manifest_reader_without_wal_store() { + let object_store: Arc = Arc::new(InMemory::new()); + let path = Path::from("/manifest-reader-without-table-store"); + let manifest_reader: Arc = + Arc::new(ManifestStore::new(&path, object_store)); + + assert_invalid_build( + SlateDbWalReaderBuilder::new().with_manifest_reader(manifest_reader), + "must specify object store", + ); + } + + #[test] + fn builder_accepts_wal_store_and_manifest_reader() { + let object_store: Arc = Arc::new(InMemory::new()); + let path = Path::from("/table-store-and-manifest-reader"); + let wal_store = Arc::new(WalTableStore::new( + Arc::clone(&object_store), + SsTableFormat::default(), + path.clone(), + TableStoreKind::Reader, + )); + let manifest_reader: Arc = + Arc::new(ManifestStore::new(&path, object_store)); + + assert!(SlateDbWalReaderBuilder::new() + .with_wal_store(wal_store) + .with_manifest_reader(manifest_reader) + .build() + .is_ok()); + } + + async fn collect_batches( + wal_reader: &SlateDbWalReader, + start_wal_id: u64, + end_wal_id_exclusive: u64, + ) -> Result, WalError> { + let mut iterator = wal_reader + .iterator((start_wal_id..end_wal_id_exclusive).into()) + .await?; + let mut batches = Vec::new(); + while let Some(batch) = iterator.next().await? { + batches.push(batch); + } + Ok(batches) + } + + #[tokio::test] + async fn bounded_polling_discovers_new_wals_and_advances_the_cursor() { + let object_store: Arc = Arc::new(InMemory::new()); + let path = Path::from("/bounded_polling_discovers_new_wals"); + let db = Db::open(path.clone(), Arc::clone(&object_store)) + .await + .unwrap(); + db.put(b"first", b"value-1").await.unwrap(); + db.flush_with_options(FlushOptions { + flush_type: FlushType::Wal, + }) + .await + .unwrap(); + + let wal_reader = SlateDbWalReaderBuilder::new() + .with_object_store(Arc::clone(&object_store)) + .with_path(path) + .build() + .unwrap(); + let first_tail = wal_reader.last_wal_file_id(0).await.unwrap(); + let first_batches = collect_batches(&wal_reader, 1, end_after(first_tail)) + .await + .unwrap(); + let first_cursors: Vec<_> = first_batches + .iter() + .map(|batch| batch.last_consumed_wal_file_id) + .collect(); + assert_eq!(first_cursors, (1..=first_tail).collect::>()); + let first_rows: Vec<_> = first_batches + .iter() + .flat_map(|batch| batch.rows.iter()) + .collect(); + assert_eq!(first_rows.len(), 1); + assert_eq!(first_rows[0].key.as_ref(), b"first"); + + let mut cursor = first_batches.last().unwrap().last_consumed_wal_file_id; + assert_eq!(cursor, first_tail); + assert_eq!(wal_reader.last_wal_file_id(cursor).await.unwrap(), cursor); + + db.put(b"second", b"value-2").await.unwrap(); + db.flush_with_options(FlushOptions { + flush_type: FlushType::Wal, + }) + .await + .unwrap(); + + let second_tail = wal_reader.last_wal_file_id(cursor).await.unwrap(); + assert!(second_tail > cursor); + let second_batches = collect_batches( + &wal_reader, + cursor.checked_add(1).unwrap(), + end_after(second_tail), + ) + .await + .unwrap(); + let second_rows: Vec<_> = second_batches + .iter() + .flat_map(|batch| batch.rows.iter()) + .collect(); + assert_eq!(second_rows.len(), 1); + assert_eq!(second_rows[0].key.as_ref(), b"second"); + cursor = second_batches.last().unwrap().last_consumed_wal_file_id; + assert_eq!(cursor, second_tail); + assert_eq!(wal_reader.last_wal_file_id(cursor).await.unwrap(), cursor); + } + + #[tokio::test] + async fn bounded_iteration_preserves_value_tombstone_merge_and_sequence_order() { + let object_store: Arc = Arc::new(InMemory::new()); + let path = Path::from("/bounded_iteration_preserves_row_kinds"); + let db = Db::builder(path.clone(), Arc::clone(&object_store)) + .with_merge_operator(Arc::new(StringConcatMergeOperator)) + .build() + .await + .unwrap(); + + db.put(b"a", b"1").await.unwrap(); + db.put(b"b", b"2").await.unwrap(); + db.flush_with_options(FlushOptions { + flush_type: FlushType::Wal, + }) + .await + .unwrap(); + db.delete(b"a").await.unwrap(); + db.flush_with_options(FlushOptions { + flush_type: FlushType::Wal, + }) + .await + .unwrap(); + db.merge(b"m", b"x").await.unwrap(); + db.flush_with_options(FlushOptions { + flush_type: FlushType::Wal, + }) + .await + .unwrap(); + + let wal_reader = SlateDbWalReaderBuilder::new() + .with_object_store(object_store) + .with_path(path) + .build() + .unwrap(); + let tail = wal_reader.last_wal_file_id(0).await.unwrap(); + let batches = collect_batches(&wal_reader, 1, end_after(tail)) + .await + .unwrap(); + assert_eq!(batches.last().unwrap().last_consumed_wal_file_id, tail); + let rows: Vec<_> = batches.into_iter().flat_map(|batch| batch.rows).collect(); + assert_eq!(rows.len(), 4); + assert!(rows.windows(2).all(|pair| pair[0].seq < pair[1].seq)); + assert_eq!(rows[0].key.as_ref(), b"a"); + assert!(matches!( + &rows[0].value, + ValueDeletable::Value(value) if value.as_ref() == b"1" + )); + assert_eq!(rows[1].key.as_ref(), b"b"); + assert!(matches!( + &rows[1].value, + ValueDeletable::Value(value) if value.as_ref() == b"2" + )); + assert_eq!(rows[2].key.as_ref(), b"a"); + assert!(matches!(rows[2].value, ValueDeletable::Tombstone)); + assert_eq!(rows[3].key.as_ref(), b"m"); + assert!(matches!( + &rows[3].value, + ValueDeletable::Merge(value) if value.as_ref() == b"x" + )); + } + + #[tokio::test] + async fn empty_fence_wal_advances_the_cursor() { + let object_store: Arc = Arc::new(InMemory::new()); + let path = Path::from("/empty_fence_wal_advances_the_cursor"); + let _db = Db::open(path.clone(), Arc::clone(&object_store)) + .await + .unwrap(); + + let wal_reader = SlateDbWalReaderBuilder::new() + .with_object_store(object_store) + .with_path(path) + .build() + .unwrap(); + let tail = wal_reader.last_wal_file_id(0).await.unwrap(); + let batches = collect_batches(&wal_reader, 1, end_after(tail)) + .await + .unwrap(); + + assert_eq!(tail, 1); + assert_eq!(batches.len(), 1); + assert!(batches[0].rows.is_empty()); + assert_eq!(batches[0].last_consumed_wal_file_id, 1); + } + + #[tokio::test] + async fn should_accept_an_unbounded_end_range() { + let object_store: Arc = Arc::new(InMemory::new()); + let path = Path::from("/reader_accepts_an_unbounded_end_range"); + let _db = Db::open(path.clone(), Arc::clone(&object_store)) + .await + .unwrap(); + let wal_reader = SlateDbWalReaderBuilder::new() + .with_object_store(object_store) + .with_path(path) + .build() + .unwrap(); + let mut iterator = wal_reader.iterator((1..).into()).await.unwrap(); + + let first = tokio::time::timeout(Duration::from_secs(1), iterator.next()) + .await + .expect("unbounded iterator did not observe the fence WAL") + .unwrap() + .expect("unbounded iterator returned None"); + assert!(first.rows.is_empty()); + assert_eq!(first.last_consumed_wal_file_id, 1); + } + + #[tokio::test] + async fn should_reject_an_unbounded_start_range() { + let object_store: Arc = Arc::new(InMemory::new()); + let wal_reader = SlateDbWalReaderBuilder::new() + .with_object_store(object_store) + .with_path(Path::from("/reader_rejects_an_unbounded_start_range")) + .build() + .unwrap(); + + let result = wal_reader + .iterator(WalFileRange(Bound::Unbounded, Bound::Unbounded)) + .await; + assert!(result.is_err()); + } + + #[tokio::test] + async fn should_normalize_excluded_start_and_included_end_bounds() { + let object_store: Arc = Arc::new(InMemory::new()); + let wal_reader = SlateDbWalReaderBuilder::new() + .with_object_store(object_store) + .with_path(Path::from("/reader_normalizes_range_bounds")) + .build() + .unwrap(); + wal_reader.wal_store.write_wal_fence(1).await.unwrap(); + wal_reader.wal_store.write_wal_fence(2).await.unwrap(); + + let mut iterator = wal_reader + .iterator(WalFileRange(Bound::Excluded(1), Bound::Included(2))) + .await + .unwrap(); + let batch = iterator.next().await.unwrap().unwrap(); + assert_eq!(batch.last_consumed_wal_file_id, 2); + assert!(batch.rows.is_empty()); + assert!(iterator.next().await.unwrap().is_none()); + } + + #[tokio::test] + async fn reads_manifest_and_wals_from_separate_object_stores() { + let object_store: Arc = Arc::new(InMemory::new()); + let wal_object_store: Arc = Arc::new(InMemory::new()); + let path = Path::from("/reader_with_dedicated_wal_store"); + let db = Db::builder(path.clone(), Arc::clone(&object_store)) + .with_wal_object_store(Arc::clone(&wal_object_store)) + .build() + .await + .unwrap(); + db.put(b"dedicated", b"wal-store").await.unwrap(); + db.flush_with_options(FlushOptions { + flush_type: FlushType::Wal, + }) + .await + .unwrap(); + + let wal_reader = SlateDbWalReaderBuilder::new() + .with_object_store(object_store) + .with_wal_object_store(wal_object_store) + .with_path(path) + .build() + .unwrap(); + let tail = wal_reader.last_wal_file_id(0).await.unwrap(); + let rows: Vec<_> = collect_batches(&wal_reader, 1, end_after(tail)) + .await + .unwrap() + .into_iter() + .flat_map(|batch| batch.rows) + .collect(); + + assert_eq!(rows.len(), 1); + assert_eq!(rows[0].key.as_ref(), b"dedicated"); + assert!(matches!( + &rows[0].value, + ValueDeletable::Value(value) if value.as_ref() == b"wal-store" + )); + } + + #[tokio::test] + async fn missing_wal_in_a_bounded_range_returns_truncation() { + let object_store: Arc = Arc::new(InMemory::new()); + let path = Path::from("/missing_wal_in_a_bounded_range"); + let db = Db::open(path.clone(), Arc::clone(&object_store)) + .await + .unwrap(); + db.put(b"key", b"value").await.unwrap(); + db.flush_with_options(FlushOptions { + flush_type: FlushType::Wal, + }) + .await + .unwrap(); + + let wal_reader = SlateDbWalReaderBuilder::new() + .with_object_store(Arc::clone(&object_store)) + .with_path(path.clone()) + .build() + .unwrap(); + let tail = wal_reader.last_wal_file_id(0).await.unwrap(); + let wal_path = PathResolver::from_root(path).sst_path(&SsTableId::Wal(tail)); + object_store.delete(&wal_path).await.unwrap(); + + let mut iterator = wal_reader + .iterator((tail..end_after(tail)).into()) + .await + .unwrap(); + assert!(matches!( + iterator.next().await, + Err(WalError::WalTruncated(wal_id)) if wal_id == tail + )); + } + + #[tokio::test] + async fn last_wal_file_id_errors_when_last_id_precedes_manifest_gc_cutoff() { + let object_store: Arc = Arc::new(InMemory::new()); + let path = Path::from("/last_wal_file_id_before_gc_cutoff"); + let wal_store = Arc::new(WalTableStore::new( + Arc::clone(&object_store), + SsTableFormat::default(), + path.clone(), + TableStoreKind::Reader, + )); + let encoded_sst = wal_store.table_builder().build().await.unwrap(); + wal_store.write_sst(1, &encoded_sst).await.unwrap(); + + let mut core = ManifestCore::new(); + core.next_wal_sst_id = 3; + core.replay_after_wal_id = 2; + StoredManifest::create_new_db( + Arc::new(ManifestStore::new(&path, Arc::clone(&object_store))), + core, + Arc::new(DefaultSystemClock::new()), + ) + .await + .unwrap(); + + let wal_reader = SlateDbWalReaderBuilder::new() + .with_object_store(object_store) + .with_path(path) + .build() + .unwrap(); + assert!(matches!( + wal_reader.last_wal_file_id(0).await, + Err(WalError::WalTruncated(1)) + )); + } +} diff --git a/slatedb/src/wal/wal_sst_builder.rs b/slatedb/src/wal/slatedb/sst_builder.rs similarity index 99% rename from slatedb/src/wal/wal_sst_builder.rs rename to slatedb/src/wal/slatedb/sst_builder.rs index 595d2c0bf4..3f98357304 100644 --- a/slatedb/src/wal/wal_sst_builder.rs +++ b/slatedb/src/wal/slatedb/sst_builder.rs @@ -868,7 +868,7 @@ mod tests { let encoded = builder.build().await.unwrap(); // then: - assert_eq!(encoded.info.sst_type, crate::db_state::SstType::Wal,); + assert_eq!(encoded.info.sst_type, SstType::Wal,); } mod block_transformer_tests { diff --git a/slatedb/src/wal/slatedb/sst_iterator.rs b/slatedb/src/wal/slatedb/sst_iterator.rs new file mode 100644 index 0000000000..5e41c56481 --- /dev/null +++ b/slatedb/src/wal/slatedb/sst_iterator.rs @@ -0,0 +1,171 @@ +#![allow(dead_code)] // Implemented ahead of migrating WAL replay to WalTableStore. + +use std::collections::VecDeque; +use std::sync::Arc; + +use super::store::{WalFileHandle, WalTableStore}; +use crate::block_iterator::DataBlockIterator; +use crate::config::SstBlockSize; +use crate::error::SlateDBError; +use crate::flatbuffer_types::SsTableIndexOwned; +use crate::format::block::Block; +use crate::iter::IterationOrder; +use crate::types::RowEntry; + +#[derive(Clone, Debug)] +pub(crate) struct WalSstIteratorOptions { + /// Target encoded bytes per block-fetch request. + pub(crate) target_bytes_to_fetch: usize, +} + +impl Default for WalSstIteratorOptions { + fn default() -> Self { + Self { + target_bytes_to_fetch: SstBlockSize::default().as_bytes(), + } + } +} + +/// Iterates all rows in one WAL SST in ascending sequence order. +/// +/// WAL replay only performs whole-file scans, so this iterator deliberately has +/// no range/view abstraction, filters, cache controls, descending mode, seek, +/// or speculative fetch scheduling. +pub(crate) struct WalSstIterator { + table: WalFileHandle, + index: Arc, + block_iter: Option>>, + next_block_idx_to_fetch: usize, + fetched_blocks: VecDeque>, + table_store: Arc, + options: WalSstIteratorOptions, +} + +impl WalSstIterator { + pub(crate) async fn new( + table: WalFileHandle, + table_store: Arc, + options: WalSstIteratorOptions, + ) -> Result { + assert!(options.target_bytes_to_fetch > 0); + let index = table_store.read_index(&table).await?; + Ok(Self { + table, + index, + block_iter: None, + next_block_idx_to_fetch: 0, + fetched_blocks: VecDeque::new(), + table_store, + options, + }) + } + + /// Returns the next WAL row in ascending sequence order. + pub(crate) async fn next(&mut self) -> Result, SlateDBError> { + loop { + if let Some(iter) = &mut self.block_iter { + if let Some(row) = iter.next().await? { + return Ok(Some(row)); + } + } + if !self.load_next_block().await? { + return Ok(None); + } + } + } + + async fn load_next_block(&mut self) -> Result { + loop { + if let Some(block) = self.fetched_blocks.pop_front() { + self.block_iter = Some(DataBlockIterator::new( + block, + self.table.format_version, + IterationOrder::Ascending, + )?); + return Ok(true); + } + + let num_blocks = self.index.borrow().block_meta().len(); + if self.next_block_idx_to_fetch == num_blocks { + self.block_iter = None; + return Ok(false); + } + + let blocks = self.table_store.block_range_for_target_bytes( + &self.table, + &self.index, + self.next_block_idx_to_fetch, + self.options.target_bytes_to_fetch, + ); + let next_block_idx_to_fetch = blocks.end; + let fetched_blocks = self + .table_store + .read_blocks_using_index(&self.table, Arc::clone(&self.index), blocks) + .await?; + // Commit the cursor only after the read succeeds so cancelling the + // read future cannot cause the next call to skip these blocks. + self.next_block_idx_to_fetch = next_block_idx_to_fetch; + self.fetched_blocks = fetched_blocks; + } + } +} + +#[cfg(test)] +mod tests { + use object_store::memory::InMemory; + use object_store::path::Path; + use object_store::ObjectStore; + + use super::*; + use crate::flatbuffer_types::FlatBufferSsTableInfoCodec; + use crate::format::sst::SsTableFormat; + use crate::object_store_tag::TableStoreKind; + use crate::wal::slatedb::sst_builder::EncodedWalSsTableBuilder; + + fn test_store() -> Arc { + let object_store: Arc = Arc::new(InMemory::new()); + Arc::new(WalTableStore::new( + object_store, + SsTableFormat::default(), + Path::from("test-db"), + TableStoreKind::Main, + )) + } + + #[tokio::test] + async fn should_iterate_the_whole_wal_sequentially() { + let store = test_store(); + let rows: Vec<_> = (1..=6) + .map(|seq| { + RowEntry::new_value( + format!("key-{seq}").as_bytes(), + format!("value-{seq}").as_bytes(), + seq, + ) + }) + .collect(); + let mut builder = + EncodedWalSsTableBuilder::new(32, Box::new(FlatBufferSsTableInfoCodec {})); + for row in rows.iter().cloned() { + builder.add(row).await.unwrap(); + } + let encoded = builder.build().await.unwrap(); + let table = store.write_sst(1, &encoded).await.unwrap(); + let mut iter = WalSstIterator::new( + table, + store, + WalSstIteratorOptions { + target_bytes_to_fetch: 1, + }, + ) + .await + .unwrap(); + + let mut actual = Vec::new(); + while let Some(row) = iter.next().await.unwrap() { + actual.push(row); + } + assert_eq!(actual, rows); + assert!(iter.next().await.unwrap().is_none()); + } +} diff --git a/slatedb/src/wal/slatedb/store.rs b/slatedb/src/wal/slatedb/store.rs new file mode 100644 index 0000000000..719f4684fe --- /dev/null +++ b/slatedb/src/wal/slatedb/store.rs @@ -0,0 +1,615 @@ +#![allow(dead_code)] // This store is intentionally implemented before its call sites are migrated. + +use std::collections::VecDeque; +use std::mem::size_of; +use std::ops::Range; +use std::sync::Arc; + +use bytes::Bytes; +use fail_parallel::{fail_point, FailPointRegistry}; +use futures::{future::join_all, StreamExt}; +use log::debug; +use object_store::path::Path; +use object_store::{ObjectStore, ObjectStoreExt, PutMode, PutOptions}; +use serde::Serialize; +use slatedb_common::object_metadata::IdentifiedObjectMetadata; + +use crate::db_state::{SsTableId, SsTableInfo, SstType}; +use crate::error::SlateDBError; +use crate::flatbuffer_types::SsTableIndexOwned; +use crate::format::block::Block; +use crate::format::sst::{ + EncodedSsTable, SsTableFormat, CHECKSUM_SIZE, METADATA_OFFSET_SIZE, VERSION_SIZE, +}; +use crate::object_store_tag::{ObjectStoreCallTag, TableStoreKind}; +use crate::paths::PathResolver; +use crate::sst_io::{read_obj, read_with_validation_retry, ReadOnlyObject}; +use crate::wal::slatedb::sst_builder::EncodedWalSsTableBuilder; + +#[derive(Clone, Copy, Debug, Eq, Hash, Ord, PartialEq, PartialOrd, Serialize)] +pub(crate) struct WalFileId(u64); + +impl WalFileId { + pub(crate) fn value(self) -> u64 { + self.0 + } +} + +impl From for WalFileId { + fn from(value: u64) -> Self { + Self(value) + } +} + +#[derive(Clone, Debug, PartialEq, Serialize)] +pub(crate) struct WalFileHandle { + pub(crate) id: WalFileId, + pub(crate) format_version: u16, + pub(crate) info: SsTableInfo, +} + +impl WalFileHandle { + fn new(id: u64, format_version: u16, info: SsTableInfo) -> Self { + Self { + id: id.into(), + format_version, + info, + } + } +} + +/// Cacheless storage adapter for SlateDB's object-store-backed WAL files. +/// +/// WALs continue to use the shared SST format and builders. This type owns only +/// WAL object-store operations and deliberately has no `DbCache` integration. +pub(crate) struct WalTableStore { + object_store: Arc, + sst_format: SsTableFormat, + path_resolver: PathResolver, + fp_registry: Arc, + kind: TableStoreKind, +} + +impl WalTableStore { + pub(crate) fn new>( + object_store: Arc, + sst_format: SsTableFormat, + root_path: P, + kind: TableStoreKind, + ) -> Self { + Self::new_with_fp_registry( + object_store, + sst_format, + PathResolver::from_root(root_path), + Arc::new(FailPointRegistry::new()), + kind, + ) + } + + pub(crate) fn new_with_fp_registry( + object_store: Arc, + sst_format: SsTableFormat, + path_resolver: PathResolver, + fp_registry: Arc, + kind: TableStoreKind, + ) -> Self { + Self { + object_store, + sst_format, + path_resolver, + fp_registry, + kind, + } + } + + pub(crate) fn table_builder(&self) -> EncodedWalSsTableBuilder { + self.sst_format.wal_table_builder() + } + + #[cfg(test)] + pub(crate) fn wal_table_builder(&self) -> EncodedWalSsTableBuilder { + self.table_builder() + } + + pub(crate) fn estimate_encoded_size(&self, num_entries: usize, size_entries: usize) -> usize { + self.sst_format + .estimate_encoded_size_wal(num_entries, size_entries) + } + + pub(crate) fn validate_wal_sst_replay_memory( + &self, + encoded_sst: &EncodedSsTable, + metadata_memory_limit: usize, + block_memory_limit: usize, + ) -> Result<(), SlateDBError> { + let index_encoded_bytes = usize::try_from(encoded_sst.info.index_len).map_err(|_| { + SlateDBError::WalReplayMemoryLimitExceeded { + kind: "encoded and decoded WAL index", + required_bytes: usize::MAX, + limit_bytes: metadata_memory_limit, + } + })?; + let metadata_encoded_bytes = encoded_sst + .footer + .len() + .checked_sub(METADATA_OFFSET_SIZE + VERSION_SIZE) + .and_then(|footer_bytes| footer_bytes.checked_sub(index_encoded_bytes)) + .ok_or(SlateDBError::InvalidDBState)?; + let retained_info_bytes = encoded_sst + .info + .first_entry + .as_ref() + .map_or(0, Bytes::len) + .checked_add(encoded_sst.info.last_entry.as_ref().map_or(0, Bytes::len)) + .ok_or(SlateDBError::WalReplayMemoryLimitExceeded { + kind: "encoded and decoded WAL metadata", + required_bytes: usize::MAX, + limit_bytes: metadata_memory_limit, + })?; + let metadata_required_bytes = metadata_encoded_bytes + .checked_add(retained_info_bytes) + .ok_or(SlateDBError::WalReplayMemoryLimitExceeded { + kind: "encoded and decoded WAL metadata", + required_bytes: usize::MAX, + limit_bytes: metadata_memory_limit, + })?; + if metadata_required_bytes > metadata_memory_limit { + return Err(SlateDBError::WalReplayMemoryLimitExceeded { + kind: "encoded and decoded WAL metadata", + required_bytes: metadata_required_bytes, + limit_bytes: metadata_memory_limit, + }); + } + + let index_payload_bytes = index_encoded_bytes + .checked_sub(CHECKSUM_SIZE) + .ok_or(SlateDBError::InvalidDBState)?; + let transformed_index_bytes = match &self.sst_format.block_transformer { + Some(transformer) => transformer.max_decoded_len(index_payload_bytes).ok_or( + SlateDBError::WalReplayMemoryLimitExceeded { + kind: "transformed WAL index", + required_bytes: usize::MAX, + limit_bytes: metadata_memory_limit, + }, + )?, + None => 0, + }; + let index_required_bytes = retained_info_bytes + .checked_add(index_encoded_bytes) + .and_then(|bytes| bytes.checked_add(transformed_index_bytes)) + .and_then(|bytes| bytes.checked_add(encoded_sst.index.size())) + .ok_or(SlateDBError::WalReplayMemoryLimitExceeded { + kind: "encoded and decoded WAL index", + required_bytes: usize::MAX, + limit_bytes: metadata_memory_limit, + })?; + if index_required_bytes > metadata_memory_limit { + return Err(SlateDBError::WalReplayMemoryLimitExceeded { + kind: "encoded and decoded WAL index", + required_bytes: index_required_bytes, + limit_bytes: metadata_memory_limit, + }); + } + + for block in &encoded_sst.unconsumed_blocks { + let encoded_payload_bytes = block + .encoded_bytes + .len() + .checked_sub(CHECKSUM_SIZE) + .ok_or(SlateDBError::InvalidDBState)?; + let transformed_block_bytes = match &self.sst_format.block_transformer { + Some(transformer) => transformer.max_decoded_len(encoded_payload_bytes).ok_or( + SlateDBError::WalReplayMemoryLimitExceeded { + kind: "transformed WAL block", + required_bytes: usize::MAX, + limit_bytes: block_memory_limit, + }, + )?, + None => 0, + }; + let offsets_bytes = block + .block + .offsets + .len() + .checked_mul(size_of::()) + .ok_or(SlateDBError::WalReplayMemoryLimitExceeded { + kind: "encoded and decoded WAL block", + required_bytes: usize::MAX, + limit_bytes: block_memory_limit, + })?; + let required_bytes = block + .encoded_bytes + .len() + .checked_add(transformed_block_bytes) + .and_then(|bytes| bytes.checked_add(block.block.size())) + .and_then(|bytes| bytes.checked_add(offsets_bytes)) + .ok_or(SlateDBError::WalReplayMemoryLimitExceeded { + kind: "encoded and decoded WAL block", + required_bytes: usize::MAX, + limit_bytes: block_memory_limit, + })?; + if required_bytes > block_memory_limit { + return Err(SlateDBError::WalReplayMemoryLimitExceeded { + kind: "encoded and decoded WAL block", + required_bytes, + limit_bytes: block_memory_limit, + }); + } + } + Ok(()) + } + + pub(crate) async fn list_wal_ssts_for_replay( + &self, + id_range: Range, + ) -> Result>, SlateDBError> { + if id_range.is_empty() { + return Ok(Vec::new()); + } + + let wal_path = self.path_resolver.wal_path(); + let mut files = if id_range.start == 0 { + self.object_store.list(Some(&wal_path)) + } else { + let offset = self.path(id_range.start - 1); + self.object_store.list_with_offset(Some(&wal_path), &offset) + }; + let mut wal_files = Vec::new(); + while let Some(file) = files.next().await.transpose()? { + let Ok(Some(SsTableId::Wal(wal_id))) = + self.path_resolver.parse_table_id(&file.location) + else { + continue; + }; + if wal_id >= id_range.end { + break; + } + if wal_id >= id_range.start { + wal_files.push(IdentifiedObjectMetadata::from_object_meta( + SsTableId::Wal(wal_id), + file, + )); + } + } + wal_files.sort_by_key(|metadata| metadata.id.unwrap_wal_id()); + Ok(wal_files) + } + + /// Writes a WAL SST with create-if-absent semantics required for fencing. + pub(crate) async fn write_sst( + &self, + wal_id: u64, + encoded_sst: &EncodedSsTable, + ) -> Result { + fail_point!(self.fp_registry.clone(), "write-wal-sst-io-error", |_| { + Err(slatedb_io_error()) + }); + + self.write_create(wal_id, encoded_sst.remaining_as_bytes()) + .await?; + Ok(WalFileHandle::new( + wal_id, + encoded_sst.format_version, + encoded_sst.info.clone(), + )) + } + + /// Writes a zero-byte WAL object as a fencing marker. + pub(crate) async fn write_wal_fence(&self, wal_id: u64) -> Result<(), SlateDBError> { + fail_point!(self.fp_registry.clone(), "write-wal-sst-io-error", |_| { + Err(slatedb_io_error()) + }); + self.write_create(wal_id, Bytes::new()).await + } + + async fn write_create(&self, wal_id: u64, data: Bytes) -> Result<(), SlateDBError> { + let path = self.path(wal_id); + let opts = PutOptions { + mode: PutMode::Create, + extensions: ObjectStoreCallTag::new(self.kind, SstType::Wal).into(), + ..PutOptions::default() + }; + self.object_store + .put_opts(&path, data.into(), opts) + .await + .map_err(|error| match error { + object_store::Error::AlreadyExists { .. } => { + debug!("path already exists [path={}]", path); + SlateDBError::Fenced + } + error => SlateDBError::from(error), + })?; + Ok(()) + } + + pub(crate) async fn open_sst(&self, wal_id: u64) -> Result { + let (info, version) = read_obj!( + Arc::clone(&self.object_store), + self.path(wal_id), + ObjectStoreCallTag::new(self.kind, SstType::Wal), + |obj| self.sst_format.read_info_and_version(&obj) + ) + .await?; + Ok(WalFileHandle::new(wal_id, version, info)) + } + + pub(crate) async fn read_index( + &self, + handle: &WalFileHandle, + ) -> Result, SlateDBError> { + let index = read_obj!( + Arc::clone(&self.object_store), + self.path(handle.id.value()), + ObjectStoreCallTag::new(self.kind, SstType::Wal), + |obj| self.sst_format.read_index(&handle.info, &obj) + ) + .await?; + Ok(Arc::new(index)) + } + + pub(crate) fn block_range_for_target_bytes( + &self, + handle: &WalFileHandle, + index: &SsTableIndexOwned, + first_block: usize, + target_bytes: usize, + ) -> Range { + assert!(target_bytes > 0); + + let index = index.borrow(); + let block_meta = index.block_meta(); + let num_blocks = block_meta.len(); + assert!(first_block < num_blocks); + let target_bytes = u64::try_from(target_bytes).unwrap_or(u64::MAX); + + let mut blocks = first_block..first_block + 1; + loop { + let byte_range = self + .sst_format + .block_range(blocks.clone(), &handle.info, &index); + if byte_range.end.saturating_sub(byte_range.start) >= target_bytes + || blocks.end == num_blocks + { + return blocks; + } + blocks.end += 1; + } + } + + pub(crate) fn block_range_size( + &self, + handle: &WalFileHandle, + index: &SsTableIndexOwned, + blocks: Range, + ) -> usize { + if blocks.is_empty() { + return 0; + } + let byte_range = self + .sst_format + .block_range(blocks, &handle.info, &index.borrow()); + usize::try_from(byte_range.end.saturating_sub(byte_range.start)).unwrap_or(usize::MAX) + } + + pub(crate) async fn read_blocks_using_index( + &self, + handle: &WalFileHandle, + index: Arc, + blocks: Range, + ) -> Result>, SlateDBError> { + let object_store = Arc::clone(&self.object_store); + let path = self.path(handle.id.value()); + let index = &index; + let blocks = + read_with_validation_retry(ObjectStoreCallTag::new(self.kind, SstType::Wal), |tag| { + let obj = ReadOnlyObject { + object_store: Arc::clone(&object_store), + path: path.clone(), + tag, + }; + let blocks = blocks.clone(); + async move { + self.sst_format + .read_blocks(&handle.info, index, blocks, &obj) + .await + .map_err(|error| error.with_path(&obj.path)) + } + }) + .await?; + Ok(blocks.into_iter().map(Arc::new).collect()) + } + + /// Find the highest WAL SST id present in the object store at or above + /// `start_after + 1`, returning `start_after` if none exist. + /// + /// `start_after` should be a known lower bound (e.g. `replay_after_wal_id` + /// from the manifest, or the highest already-replayed WAL id). Passing 0 + /// scans the entire WAL id space. + /// + /// Two phases: + /// 1. Parallel exponential probe at offsets `2^0, 2^1, ..., 2^k` from + /// `start_after`. One RTT per round of 8 exponents. Brackets the + /// frontier between two adjacent powers of two. + /// 2. Sequential binary search inside the bracketed range to find the + /// exact frontier. + /// + /// Relies on the fencing protocol's contiguity invariant: "id exists" is + /// monotone-decreasing in id, so binary search is sound. Total HEAD count + /// is `O(log N)` for a gap of size N, vs `O(N)` for a windowed scan. + pub(crate) async fn last_seen_wal_id(&self, start_after: u64) -> Result { + fail_point!(Arc::clone(&self.fp_registry), "probe-wal-ssts", |_| { + Err(SlateDBError::from(std::io::Error::other("oops"))) + }); + + const ROUND_SIZE: u32 = 8; + const MAX_EXP: u32 = 48; + + let mut lo_offset = None; + let mut hi_offset = None; + let mut next_exp = 0; + + while hi_offset.is_none() { + if next_exp >= MAX_EXP { + return Err(SlateDBError::InvalidDBState); + } + let end_exp = (next_exp + ROUND_SIZE).min(MAX_EXP); + let exps: Vec = (next_exp..end_exp).collect(); + let probes = exps.iter().map(|&exp| { + let offset = 1u64 << exp; + let path = self.path(start_after + offset); + let object_store = Arc::clone(&self.object_store); + async move { wal_object_exists(&object_store, &path).await } + }); + let results = join_all(probes).await; + + for (exp, result) in exps.iter().zip(results) { + let offset = 1u64 << exp; + if result? { + lo_offset = Some(offset); + } else { + hi_offset = Some(offset); + break; + } + } + next_exp = end_exp; + } + + let hi = hi_offset.expect("loop exits only after finding an upper bound"); + let Some(lo) = lo_offset else { + return Ok(start_after); + }; + + let mut left = lo + 1; + let mut right = hi; + while left < right { + let mid = left + (right - left) / 2; + if wal_object_exists(&self.object_store, &self.path(start_after + mid)).await? { + left = mid + 1; + } else { + right = mid; + } + } + Ok(start_after + left - 1) + } + + pub(crate) async fn next_wal_sst_id( + &self, + wal_id_last_compacted: u64, + ) -> Result { + Ok(self.last_seen_wal_id(wal_id_last_compacted).await? + 1) + } + + fn path(&self, wal_id: u64) -> Path { + self.path_resolver.sst_path(&SsTableId::Wal(wal_id)) + } +} + +async fn wal_object_exists( + object_store: &Arc, + path: &Path, +) -> Result { + match object_store.head(path).await { + Ok(_) => Ok(true), + Err(object_store::Error::NotFound { .. }) => Ok(false), + Err(error) => Err(SlateDBError::from(error)), + } +} + +fn slatedb_io_error() -> SlateDBError { + SlateDBError::from(std::io::Error::other("oops")) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::block_iterator::DataBlockIterator; + use crate::iter::IterationOrder; + use crate::types::RowEntry; + use object_store::memory::InMemory; + + fn test_store() -> WalTableStore { + let object_store: Arc = Arc::new(InMemory::new()); + WalTableStore::new( + object_store, + SsTableFormat::default(), + Path::from("test-db"), + TableStoreKind::Main, + ) + } + + #[tokio::test] + async fn writes_and_reads_wal_sst_without_a_cache() { + let store = test_store(); + let rows = [ + RowEntry::new_value(b"first", b"value-1", 10), + RowEntry::new_value(b"second", b"value-2", 11), + ]; + assert!(store.estimate_encoded_size(2, 26) > 0); + + let mut builder = store.table_builder(); + for row in rows.iter().cloned() { + builder.add(row).await.unwrap(); + } + let encoded = builder.build().await.unwrap(); + let written = store.write_sst(1, &encoded).await.unwrap(); + let opened = store.open_sst(1).await.unwrap(); + + assert_eq!(written, opened); + assert_eq!(opened.id.value(), 1); + + let index = store.read_index(&opened).await.unwrap(); + let block_count = index.borrow().block_meta().len(); + assert_eq!(block_count, 1); + assert_eq!( + index.borrow().block_meta().get(0).first_key().bytes(), + &10u64.to_be_bytes() + ); + + let block_range = store.block_range_for_target_bytes(&opened, &index, 0, usize::MAX); + assert_eq!(block_range, 0..block_count); + assert!(store.block_range_size(&opened, &index, block_range.clone()) > 0); + + let blocks = store + .read_blocks_using_index(&opened, Arc::clone(&index), block_range) + .await + .unwrap(); + let mut actual = Vec::new(); + for block in blocks { + let mut iter = + DataBlockIterator::new(block, opened.format_version, IterationOrder::Ascending) + .unwrap(); + while let Some(row) = iter.next().await.unwrap() { + actual.push(row); + } + } + assert_eq!(actual, rows); + } + + #[tokio::test] + async fn uses_create_semantics_for_wals_and_fences() { + let store = test_store(); + + store.write_wal_fence(1).await.unwrap(); + assert_eq!(store.last_seen_wal_id(0).await.unwrap(), 1); + assert_eq!(store.next_wal_sst_id(0).await.unwrap(), 2); + assert!(matches!( + store.write_wal_fence(1).await, + Err(SlateDBError::Fenced) + )); + assert!(matches!( + store.open_sst(1).await, + Err(SlateDBError::EmptySSTable) + )); + + let mut builder = store.table_builder(); + builder + .add(RowEntry::new_value(b"key", b"value", 12)) + .await + .unwrap(); + let encoded = builder.build().await.unwrap(); + assert!(matches!( + store.write_sst(1, &encoded).await, + Err(SlateDBError::Fenced) + )); + } +} diff --git a/slatedb/src/wal_buffer.rs b/slatedb/src/wal/slatedb/writer.rs similarity index 84% rename from slatedb/src/wal_buffer.rs rename to slatedb/src/wal/slatedb/writer.rs index 5c173a89dd..75ebddf5f0 100644 --- a/slatedb/src/wal_buffer.rs +++ b/slatedb/src/wal/slatedb/writer.rs @@ -4,16 +4,14 @@ use std::sync::atomic::{AtomicU64, Ordering}; use std::sync::Arc; use std::time::Duration; -use crate::db_state::SsTableId; +use self::stats::WalBufferStats; use crate::dispatcher::{MessageHandler, MessageHandlerExecutor, MessageTickerDef}; use crate::error::SlateDBError; -use crate::tablestore::TableStore; use crate::types::RowEntry; use crate::utils::SafeSender; -use crate::utils::{format_bytes_si, WatchableOnceCell, WatchableOnceCellReader}; +use crate::utils::{format_bytes_si, WatchableOnceCellReader}; use crate::wal; use crate::wal::{FlushResultFuture, WalError, WalEvent, WalStatus, WalWriter}; -use crate::wal_buffer_stats::WalBufferStats; use async_trait::async_trait; use futures::{stream::BoxStream, FutureExt, StreamExt}; use log::{error, trace, warn}; @@ -21,9 +19,11 @@ use slatedb_common::metrics::MetricsRecorderHelper; use tokio::{runtime::Handle, sync::oneshot}; use tracing::instrument; +use super::store::WalTableStore; + pub(crate) const WAL_BUFFER_TASK_NAME: &str = "wal_writer"; -/// [`WalBufferManager`] buffers write operations in memory before flushing them to persistent storage. +/// [`SlateDbWalWriter`] buffers write operations in memory before flushing them to persistent storage. /// The flush operation only targets Remote storage right now, later we can add an option to flush to local /// storage. /// @@ -33,8 +33,10 @@ pub(crate) const WAL_BUFFER_TASK_NAME: &str = "wal_writer"; /// /// - `max_wal_size`: Flushes when `max_wal_size` bytes is exceeded /// - `max_flush_interval`: Flushes after `max_flush_interval` elapses, if set +/// - `max_wal_flushes_before_l0_flush`: Requests a memtable flush when this many WAL files have +/// been flushed since the latest memtable replay point /// -/// For strict durability requirements on synchronous writes, use [`WalBufferManager::flush()`] to explicitly +/// For strict durability requirements on synchronous writes, use [`SlateDbWalWriter::flush()`] to explicitly /// trigger a flush operation and await the result. This will flush ALL the in memory WALs (including the /// current WAL) to remote storage. /// @@ -46,11 +48,12 @@ pub(crate) const WAL_BUFFER_TASK_NAME: &str = "wal_writer"; /// guaranteed to be written atomically to the same WAL file. /// - Fatal errors during flush operations are stored internally and propagated to all subsequent /// operations. The manager becomes unusable after encountering a fatal error. -pub(crate) struct WalBufferManager { - inner: Arc>, +pub(crate) struct SlateDbWalWriter { + inner: Arc>, stats: Arc, - table_store: Arc, + table_store: Arc, max_wal_bytes_size: usize, + max_wal_flushes_before_l0_flush: u64, max_replay_block_bytes: usize, /// The largest flush_epoch for which a size-triggered flush request has been /// sent. Compared against `flush_epoch` in the inner struct to avoid sending @@ -60,7 +63,7 @@ pub(crate) struct WalBufferManager { task_executor: Arc, } -struct WalBufferManagerInner { +struct SlateDbWalWriterInner { current_wal: WalBuffer, /// When the current WAL is ready to be flushed, it'll be moved to the `immutable_wals`. /// The flusher will try flush all the immutable wals to remote storage. @@ -93,8 +96,6 @@ struct WalBufferManagerInner { struct WalBuffer { /// queue for the entries entries: VecDeque, - /// watcher to await durability - durable: WatchableOnceCell>, /// the sequence number of the most recent addition to this WAL buffer last_seq: u64, /// size of the entries that has been added to the WAL buffer in bytes @@ -107,13 +108,14 @@ struct WalBufferIterator { entries: std::vec::IntoIter, } -impl WalBufferManager { +impl SlateDbWalWriter { pub(crate) async fn start_new( closed_result_reader: WatchableOnceCellReader>, recorder: &MetricsRecorderHelper, last_flushed_wal_id: u64, - table_store: Arc, + table_store: Arc, max_wal_bytes_size: usize, + max_wal_flushes_before_l0_flush: u64, max_replay_metadata_bytes: usize, max_replay_block_bytes: usize, max_flush_interval: Option, @@ -122,7 +124,7 @@ impl WalBufferManager { let current_wal = WalBuffer::new(); let immutable_wals = VecDeque::new(); let (flush_tx, flush_rx) = SafeSender::unbounded_channel(closed_result_reader); - let inner = WalBufferManagerInner { + let inner = SlateDbWalWriterInner { current_wal, immutable_wals, flush_epoch: 1, @@ -154,26 +156,24 @@ impl WalBufferManager { stats, table_store, max_wal_bytes_size, + max_wal_flushes_before_l0_flush, max_replay_block_bytes, last_flush_requested_epoch: AtomicU64::new(0), task_executor, }) } - //TODO: do we still need durable watchers here? /// Check if we need to flush the wal with considering max_wal_size. the checking over `max_wal_size` /// is not very strict, we have to ensure a write batch into a single WAL file. /// /// It's the caller's duty to call `maybe_trigger_flush` after calling `append`. - fn maybe_trigger_flush( - &self, - ) -> Result>, WalError> { - let (durable_watcher, need_flush, flush_epoch) = { + fn maybe_trigger_flush(&self) -> Result<(), WalError> { + let (need_flush, flush_epoch) = { let inner = self.inner.read(); // checks the size of the current wal let (need_flush, flush_epoch) = inner.needs_flush(&self.table_store, self.max_wal_bytes_size); - (inner.current_wal.durable_watcher(), need_flush, flush_epoch) + (need_flush, flush_epoch) }; if need_flush { // Only send a flush request if one hasn't already been sent for this epoch. @@ -193,7 +193,7 @@ impl WalBufferManager { self.stats .estimated_bytes .set(status.estimated_bytes as i64); - Ok(durable_watcher) + Ok(()) } /// Send a flush request to the background flush worker. @@ -209,7 +209,7 @@ impl WalBufferManager { } #[async_trait] -impl WalWriter for WalBufferManager { +impl WalWriter for SlateDbWalWriter { fn status(&self) -> Result { self.inner.read().status(&self.table_store) } @@ -240,8 +240,16 @@ impl WalWriter for WalBufferManager { Ok(()) } + fn should_flush_memtable(&self, replay_after_wal_id: u64) -> bool { + let last_flushed_wal_id = self.inner.read().last_flushed_wal_id; + let Some(wal_id_gap) = last_flushed_wal_id.checked_sub(replay_after_wal_id) else { + return false; + }; + wal_id_gap >= self.max_wal_flushes_before_l0_flush + } + fn observer(&self) -> Box { - Box::new(WalObserver { + Box::new(SlateDbWalObserver { inner: self.inner.clone(), table_store: self.table_store.clone(), }) @@ -268,12 +276,12 @@ impl WalWriter for WalBufferManager { }; self.inner .write() - .drain_on_close(WalError::Closed, &self.table_store); + .mark_closed(WalError::Closed, &self.table_store); Ok(()) } } -impl WalBufferManagerInner { +impl SlateDbWalWriterInner { fn check_exited(&self) -> Result<(), WalError> { match self.flush_task_exited_reason.as_ref() { Some(err) => Err(err.clone()), @@ -299,10 +307,10 @@ impl WalBufferManagerInner { Ok(()) } - fn needs_flush(&self, table_store: &TableStore, max_wal_bytes_size: usize) -> (bool, u64) { + fn needs_flush(&self, table_store: &WalTableStore, max_wal_bytes_size: usize) -> (bool, u64) { // check the size of the current wal let current_wal_size = - table_store.estimate_encoded_size_wal(self.current_wal.len(), self.current_wal.size()); + table_store.estimate_encoded_size(self.current_wal.len(), self.current_wal.size()); trace!( "checking flush trigger [current_wal_size={}, max_wal_bytes_size={}]", format_bytes_si(current_wal_size as u64), @@ -323,18 +331,18 @@ impl WalBufferManagerInner { } /// Returns the total size of all unflushed WALs in bytes. - fn estimated_bytes(&self, table_store: &TableStore) -> usize { + fn estimated_bytes(&self, table_store: &WalTableStore) -> usize { let current_wal_size = - table_store.estimate_encoded_size_wal(self.current_wal.len(), self.current_wal.size()); + table_store.estimate_encoded_size(self.current_wal.len(), self.current_wal.size()); let imm_wal_size = self .immutable_wals .iter() - .map(|(_, wal)| table_store.estimate_encoded_size_wal(wal.len(), wal.size())) + .map(|(_, wal)| table_store.estimate_encoded_size(wal.len(), wal.size())) .sum::(); current_wal_size + imm_wal_size } - fn status(&self, table_store: &TableStore) -> Result { + fn status(&self, table_store: &WalTableStore) -> Result { let status = self.compute_status(table_store); if status.closed_reason.is_none() { Ok(status) @@ -343,7 +351,7 @@ impl WalBufferManagerInner { } } - fn compute_status(&self, table_store: &TableStore) -> WalStatus { + fn compute_status(&self, table_store: &WalTableStore) -> WalStatus { let flushing_wal_entries_count = self .immutable_wals .iter() @@ -359,17 +367,11 @@ impl WalBufferManagerInner { } } - fn drain_on_close( - &mut self, - reason: WalError, - table_store: &TableStore, - ) -> (WalStatus, Vec<(u64, Arc)>) { + fn mark_closed(&mut self, reason: WalError, table_store: &WalTableStore) -> WalStatus { self.flush_task_exited_reason = Some(reason); self.freeze_current_wal(); - let unflushed_wals = self.flushing_wals(); self.immutable_wals.clear(); - let status = self.compute_status(table_store); - (status, unflushed_wals) + self.compute_status(table_store) } fn freeze_current_wal(&mut self) { @@ -401,7 +403,7 @@ impl WalBufferManagerInner { self.last_flushed_wal_id = flushed_wal_id; if let Some(seq) = flushed_wal.last_seq() { if let Some(last_flushed_seq) = self.last_flushed_seq { - assert!(seq >= last_flushed_seq); + assert!(seq > last_flushed_seq); } self.last_flushed_seq = Some(seq); } @@ -413,7 +415,6 @@ impl WalBuffer { fn new() -> Self { Self { entries: VecDeque::new(), - durable: WatchableOnceCell::new(), last_seq: 0, entries_size: 0, } @@ -430,22 +431,6 @@ impl WalBuffer { WalBufferIterator::new(self) } - /// Returns a watcher that can be used to await durability. - fn durable_watcher(&self) -> WatchableOnceCellReader> { - self.durable.reader() - } - - /// Awaits until the WAL is durable (flushed to storage). - #[cfg(test)] - async fn await_durable(&self) -> Result<(), SlateDBError> { - self.durable.reader().await_value().await - } - - /// Notifies that the WAL has been made durable (or failed). - fn notify_durable(&self, result: Result<(), SlateDBError>) { - self.durable.write(result); - } - /// Returns true if the buffer is empty. fn is_empty(&self) -> bool { self.entries.is_empty() @@ -508,8 +493,8 @@ struct WalFlushHandler { max_flush_interval: Option, max_replay_metadata_bytes: usize, max_replay_block_bytes: usize, - inner: Arc>, - table_store: Arc, + inner: Arc>, + table_store: Arc, stats: Arc, listener: Option, } @@ -540,18 +525,11 @@ impl WalFlushHandler { inner.compute_status(&self.table_store) }; - // we notify the listener first since that updates the oracle, and then notify - // the table waiters. blocked writes wait on the table, so we have to update the oracle - // first to preserve read-your-writes. This does mean that there is a small window - // after notifying flushed before the wal memory is actually released. - // TODO: once we change writes to block on the durable seq num from the oracle we - // can simplify this and fully drop the wal before notifying listeners - self.notify_listener(wal::WalEvent::WalFlushed(status)); - wal.notify_durable(result.clone()); if Arc::strong_count(&wal) > 1 { warn!("outstanding references to wal id {} after flushing", wal_id); } drop(wal); + self.notify_listener(WalEvent::WalFlushed(status)); } Ok(()) @@ -560,7 +538,7 @@ impl WalFlushHandler { async fn do_flush_one_wal(&self, wal_id: u64, wal: Arc) -> Result<(), SlateDBError> { self.stats.flushes.increment(1); - let mut sst_builder = self.table_store.wal_table_builder(); + let mut sst_builder = self.table_store.table_builder(); let mut iter = wal.iter(); while let Some(entry) = iter.next() { sst_builder.add(entry).await?; @@ -573,14 +551,12 @@ impl WalFlushHandler { self.max_replay_block_bytes, )?; let written_bytes = encoded_sst.remaining_len() as u64; - self.table_store - .write_sst(&SsTableId::Wal(wal_id), &encoded_sst) - .await?; + self.table_store.write_sst(wal_id, &encoded_sst).await?; self.stats.flush_bytes.increment(written_bytes); Ok(()) } - fn notify_listener(&self, event: wal::WalEvent) { + fn notify_listener(&self, event: WalEvent) { if let Some(l) = self.listener.as_ref() { (*l)(event); } @@ -630,10 +606,10 @@ impl MessageHandler for WalFlushHandler { .map(WalError::from) .unwrap_or(WalError::Closed); - let (final_status, unflushed) = self + let final_status = self .inner .write() - .drain_on_close(error.clone(), &self.table_store); + .mark_closed(error.clone(), &self.table_store); self.notify_listener(WalEvent::WalClosed(final_status.clone())); // drain remaining messages @@ -649,25 +625,18 @@ impl MessageHandler for WalFlushHandler { } } } - - // notify all the flushing wals to be finished with fatal error or shutdown - // error. we need ensure all the wal tables finally get notified. freeze current - // WAL to notify writers in the subsequent flushing_wals loop. - for (_, wal) in unflushed { - wal.notify_durable(Err(result.clone().err().unwrap_or(SlateDBError::Closed))); - } Ok(()) } } /// Interface for getting information about the current state of the Wal #[derive(Clone)] -struct WalObserver { - inner: Arc>, - table_store: Arc, +struct SlateDbWalObserver { + inner: Arc>, + table_store: Arc, } -impl wal::WalObserver for WalObserver { +impl wal::WalObserver for SlateDbWalObserver { /// Gets information about the Wal buffer's current state fn status(&self) -> Result { self.inner.read().status(self.table_store.as_ref()) @@ -717,16 +686,12 @@ pub mod stats { #[cfg(test)] mod tests { use super::*; - use crate::block_cache_policy::BlockCachePolicy; use crate::db_status::{ClosedResultWriter, DbStatusManager}; use crate::format::sst::SsTableFormat; - use crate::iter::RowEntryIterator; - use crate::manifest::SsTableView; - use crate::object_stores::ObjectStores; + use crate::object_store_tag::TableStoreKind; use crate::oracle::DbOracle; - use crate::sst_iter::{SstIterator, SstIteratorOptions}; - use crate::tablestore::{TableStore, TableStoreKind}; use crate::types::{RowEntry, ValueDeletable}; + use crate::wal::slatedb::sst_iterator::{WalSstIterator, WalSstIteratorOptions}; use bytes::Bytes; use object_store::{memory::InMemory, path::Path, ObjectStore}; use slatedb_common::clock::DefaultSystemClock; @@ -794,52 +759,6 @@ mod tests { assert_eq!(buffer.last_seq(), Some(40)); } - #[tokio::test] - async fn test_notify_durable_success() { - let mut buffer = WalBuffer::new(); - buffer.append(make_entry("key", "value", 1, None)); - - buffer.notify_durable(Ok(())); - - let result = buffer.await_durable().await; - assert!(result.is_ok()); - } - - #[tokio::test] - async fn test_notify_durable_error() { - let mut buffer = WalBuffer::new(); - buffer.append(make_entry("key", "value", 1, None)); - - buffer.notify_durable(Err(SlateDBError::Closed)); - - let result = buffer.await_durable().await; - assert!(matches!(result, Err(SlateDBError::Closed))); - } - - #[tokio::test] - async fn test_durable_watcher_returns_reader() { - let mut buffer = WalBuffer::new(); - buffer.append(make_entry("key", "value", 1, None)); - - let mut reader = buffer.durable_watcher(); - buffer.notify_durable(Ok(())); - - let result = reader.await_value().await; - assert!(result.is_ok()); - } - - #[tokio::test] - async fn test_notify_durable_only_sets_once() { - let mut buffer = WalBuffer::new(); - buffer.append(make_entry("key", "value", 1, None)); - - buffer.notify_durable(Ok(())); - buffer.notify_durable(Err(SlateDBError::Closed)); - - let result = buffer.await_durable().await; - assert!(result.is_ok()); - } - #[test] fn test_iter() { let mut buffer = WalBuffer::new(); @@ -908,8 +827,8 @@ mod tests { } async fn setup_wal_buffer() -> ( - WalBufferManager, - Arc, + SlateDbWalWriter, + Arc, Arc, Arc, ) { @@ -919,8 +838,8 @@ mod tests { async fn setup_wal_buffer_with_flush_interval( flush_interval: Duration, ) -> ( - WalBufferManager, - Arc, + SlateDbWalWriter, + Arc, Arc, Arc, ) { @@ -931,8 +850,8 @@ mod tests { flush_interval: Duration, listener: wal::WalStatusListener, ) -> ( - WalBufferManager, - Arc, + SlateDbWalWriter, + Arc, Arc, Arc, ) { @@ -951,8 +870,8 @@ mod tests { max_replay_metadata_bytes: usize, max_replay_block_bytes: usize, ) -> ( - WalBufferManager, - Arc, + SlateDbWalWriter, + Arc, Arc, Arc, ) { @@ -975,19 +894,17 @@ mod tests { max_replay_metadata_bytes: usize, max_replay_block_bytes: usize, ) -> ( - WalBufferManager, - Arc, + SlateDbWalWriter, + Arc, Arc, Arc, ) { let object_store: Arc = Arc::new(InMemory::new()); - let table_store = Arc::new(TableStore::new( - ObjectStores::new(object_store, None), + let table_store = Arc::new(WalTableStore::new( + object_store, format, Path::from("/root"), - None, TableStoreKind::Main, - BlockCachePolicy::default(), )); let system_clock = Arc::new(DefaultSystemClock::new()); let status_manager = Arc::new(DbStatusManager::new(0)); @@ -998,12 +915,13 @@ mod tests { status_manager.clone(), system_clock.clone(), )); - let wal_buffer = WalBufferManager::start_new( + let wal_buffer = SlateDbWalWriter::start_new( status_manager.result_reader(), &helper, 0, // recent_flushed_wal_id table_store.clone(), max_wal_bytes_size, + 4096, // max_wal_flushes_before_l0_flush max_replay_metadata_bytes, max_replay_block_bytes, Some(flush_interval), // max_flush_interval @@ -1015,7 +933,7 @@ mod tests { observer .subscribe(Arc::new(move |status| { (*listener)(status.clone()); - let wal::WalEvent::WalFlushed(status) = status else { + let WalEvent::WalFlushed(status) = status else { return; }; oracle.advance_durable_seq(status.last_flushed_seq.unwrap_or(0)) @@ -1027,6 +945,19 @@ mod tests { (wal_buffer, table_store, status_manager, recorder) } + #[tokio::test] + async fn test_should_flush_memtable_at_max_wal_gap() { + let (wal_buffer, _, _, _) = setup_wal_buffer().await; + let max_wal_gap = wal_buffer.max_wal_flushes_before_l0_flush; + let last_flushed_wal_id = max_wal_gap + 10; + wal_buffer.inner.write().last_flushed_wal_id = last_flushed_wal_id; + + assert!(!wal_buffer.should_flush_memtable(last_flushed_wal_id - max_wal_gap + 1)); + assert!(wal_buffer.should_flush_memtable(last_flushed_wal_id - max_wal_gap)); + assert!(wal_buffer.should_flush_memtable(last_flushed_wal_id - max_wal_gap - 1)); + assert!(!wal_buffer.should_flush_memtable(last_flushed_wal_id + 1)); + } + #[tokio::test] async fn test_basic_append_and_flush_operations() { let (mut wal_buffer, table_store, _, _) = setup_wal_buffer().await; @@ -1048,19 +979,11 @@ mod tests { wal_buffer.flush().await.unwrap().await.unwrap(); // Verify entries were written to storage - let sst_iter_options = SstIteratorOptions { - eager_spawn: true, - ..SstIteratorOptions::default() - }; - let mut iter = SstIterator::new_owned_initialized( - .., - SsTableView::identity(table_store.open_sst(&SsTableId::Wal(1)).await.unwrap()), - table_store.clone(), - sst_iter_options, - ) - .await - .unwrap() - .unwrap(); + let table = table_store.open_sst(1).await.unwrap(); + let mut iter = + WalSstIterator::new(table, table_store.clone(), WalSstIteratorOptions::default()) + .await + .unwrap(); let read_entry1 = iter.next().await.unwrap().unwrap(); assert_eq!(read_entry1.key, entry1.key); @@ -1236,8 +1159,8 @@ mod tests { ); } - fn recording_listener() -> (wal::WalStatusListener, Arc>>) { - let events = Arc::new(std::sync::Mutex::new(Vec::new())); + fn recording_listener() -> (wal::WalStatusListener, Arc>>) { + let events = Arc::new(Mutex::new(Vec::new())); let recorder = events.clone(); let listener = Arc::new(move |event| { recorder.lock().unwrap().push(event); @@ -1263,7 +1186,7 @@ mod tests { let mut flushed: Vec<_> = recorded .iter() .filter_map(|e| { - let wal::WalEvent::WalFlushed(status) = e else { + let WalEvent::WalFlushed(status) = e else { return None; }; Some(status) diff --git a/slatedb/src/wal/writer_init.rs b/slatedb/src/wal/slatedb/writer_init.rs similarity index 75% rename from slatedb/src/wal/writer_init.rs rename to slatedb/src/wal/slatedb/writer_init.rs index 050d3b45c0..5fadfbad74 100644 --- a/slatedb/src/wal/writer_init.rs +++ b/slatedb/src/wal/slatedb/writer_init.rs @@ -1,10 +1,11 @@ use crate::dispatcher::MessageHandlerExecutor; use crate::error::SlateDBError; use crate::manifest::Manifest; -use crate::tablestore::TableStore; use crate::utils::WatchableOnceCellReader; +use crate::wal::slatedb::iterator::{SlateDbWalIterator, WalIteratorEndBound}; +use crate::wal::slatedb::reader::SlateDbWalReaderOptions; +use crate::wal::slatedb::writer::SlateDbWalWriter; use crate::wal::{WalError, WriterInitResult, WriterManifest}; -use crate::wal_buffer::WalBufferManager; use crate::{wal, Settings}; use async_trait::async_trait; use fail_parallel::{fail_point_send, FailPointTx}; @@ -12,32 +13,40 @@ use slatedb_common::metrics::MetricsRecorderHelper; use std::sync::Arc; use std::time::Duration; +use super::store::WalTableStore; + #[derive(Clone, Copy)] -pub(crate) struct WalWriterInitOptions { +pub(crate) struct SlateDbWalWriterInitOptions { max_wal_bytes_size: usize, + max_wal_flushes_before_l0_flush: u64, max_replay_metadata_bytes: usize, max_replay_block_bytes: usize, + max_replay_concurrent_objects: usize, max_flush_interval: Option, } -impl From<&Settings> for WalWriterInitOptions { +impl From<&Settings> for SlateDbWalWriterInitOptions { fn from(settings: &Settings) -> Self { Self { max_wal_bytes_size: settings.l0_sst_size_bytes, + max_wal_flushes_before_l0_flush: settings.max_wal_flushes_before_l0_flush, max_replay_metadata_bytes: settings.wal_replay.metadata_working_memory_limit(), max_replay_block_bytes: settings.wal_replay.block_working_memory_limit(), + max_replay_concurrent_objects: settings.wal_replay.max_concurrent_objects, max_flush_interval: settings.flush_interval, } } } -pub(crate) struct WalWriterInit { +pub(crate) struct SlateDbWalWriterInit { closed_result_reader: WatchableOnceCellReader>, recorder: MetricsRecorderHelper, - table_store: Arc, + table_store: Arc, max_wal_bytes_size: usize, + max_wal_flushes_before_l0_flush: u64, max_replay_metadata_bytes: usize, max_replay_block_bytes: usize, + max_replay_concurrent_objects: usize, max_flush_interval: Option, empty_wal_id: u64, task_executor: Arc, @@ -45,12 +54,12 @@ pub(crate) struct WalWriterInit { fp_tx: FailPointTx, } -impl WalWriterInit { +impl SlateDbWalWriterInit { pub(crate) async fn load( closed_result_reader: WatchableOnceCellReader>, recorder: MetricsRecorderHelper, - table_store: Arc, - options: WalWriterInitOptions, + table_store: Arc, + options: SlateDbWalWriterInitOptions, manifest: &Manifest, task_executor: Arc, fp_tx: FailPointTx, @@ -64,8 +73,10 @@ impl WalWriterInit { recorder, table_store, max_wal_bytes_size: options.max_wal_bytes_size, + max_wal_flushes_before_l0_flush: options.max_wal_flushes_before_l0_flush, max_replay_metadata_bytes: options.max_replay_metadata_bytes, max_replay_block_bytes: options.max_replay_block_bytes, + max_replay_concurrent_objects: options.max_replay_concurrent_objects, max_flush_interval: options.max_flush_interval, empty_wal_id, task_executor, @@ -75,7 +86,7 @@ impl WalWriterInit { } #[async_trait] -impl wal::WriterInit for WalWriterInit { +impl wal::WriterInit for SlateDbWalWriterInit { async fn fence_and_init( &self, writer_manifest: &mut WriterManifest, @@ -117,12 +128,24 @@ impl wal::WriterInit for WalWriterInit { // older writers would have failed with a stale epoch let replay_after_wal_id = manifest.core().replay_after_wal_id; assert!(empty_wal_id > replay_after_wal_id); - let wal_writer = WalBufferManager::start_new( + let replay_iterator = SlateDbWalIterator::range( + replay_after_wal_id + 1, + WalIteratorEndBound::Exclusive(empty_wal_id + 1), + SlateDbWalReaderOptions { + sst_batch_size: self.max_replay_concurrent_objects, + read_ahead_bytes: self.max_replay_block_bytes, + ..SlateDbWalReaderOptions::default() + } + .into(), + self.table_store.clone(), + )?; + let wal_writer = SlateDbWalWriter::start_new( self.closed_result_reader.clone(), &self.recorder, empty_wal_id, self.table_store.clone(), self.max_wal_bytes_size, + self.max_wal_flushes_before_l0_flush, self.max_replay_metadata_bytes, self.max_replay_block_bytes, self.max_flush_interval, @@ -130,7 +153,7 @@ impl wal::WriterInit for WalWriterInit { ) .await?; let result = WriterInitResult { - replay_range: (replay_after_wal_id + 1..empty_wal_id + 1).into(), + replay_iterator: Box::new(replay_iterator), wal_writer: Box::new(wal_writer), }; return Ok(result); diff --git a/slatedb/src/wal_reader.rs b/slatedb/src/wal_reader.rs index cdd3eb2181..a70e41b873 100644 --- a/slatedb/src/wal_reader.rs +++ b/slatedb/src/wal_reader.rs @@ -153,8 +153,8 @@ impl WalFile { SsTableView::identity(sst), Arc::clone(&self.table_store), SstIteratorOptions { - // Optimize for throughput. Go for 256MiB per-fetch (4096 bytes/block default). - blocks_to_fetch: 65_536, + // Optimize for throughput. Go for 256MiB per fetch. + target_bytes_to_fetch: 256 * 1024 * 1024, ..Default::default() }, ) diff --git a/slatedb/src/wal_replay.rs b/slatedb/src/wal_replay.rs index 032cd8d813..d936b17e11 100644 --- a/slatedb/src/wal_replay.rs +++ b/slatedb/src/wal_replay.rs @@ -1,128 +1,34 @@ -//! WAL replay has two deliberately separate discovery and I/O paths. -//! -//! Exact recovery (writer/reader open and reader checkpoint recovery) fixes a -//! finite range with LIST, then uses concurrent full-object GETs with bounded -//! oversized-object fallback. Runtime reader replay never LISTs: it restores -//! the bounded range-read iterator and probes only explicit WAL IDs. Keeping -//! these types separate prevents recurring reader refresh from accidentally -//! inheriting exact-open discovery cost. - -use crate::block_iterator::DataBlockIterator; -use crate::config::WalReplaySettings; use crate::error::SlateDBError; -use crate::iter::{EmptyIterator, IterationOrder, RowEntryIterator}; -use crate::manifest::{ManifestCore, SsTableView}; +use crate::manifest::ManifestCore; use crate::mem_table::WritableKVTable; -use crate::replay_task_scope::ReplayTaskScope; -use crate::sst_iter::{SstIterator, SstIteratorOptions}; -use crate::tablestore::{ - DecodedWalSst, DecodedWalSstData, RangedWalSst, RangedWalSstData, RuntimeWalOpenError, - TableStore, +use crate::tablestore::TableStore; +#[cfg(test)] +use crate::wal::slatedb::iterator::{ + SlateDbWalIterator, SlateDbWalIteratorOptions, WalIteratorEndBound, }; -use crate::types::RowEntry; -use crate::utils::panic_string; -use async_trait::async_trait; -use bytes::Bytes; -use log::{error, info}; -use std::collections::VecDeque; -use std::num::NonZeroUsize; +#[cfg(test)] +use crate::wal::slatedb::store::WalTableStore; +use crate::wal::WalIterator as WalIteratorTrait; +#[cfg(test)] use std::ops::Range; use std::sync::Arc; -use tokio::sync::{OwnedSemaphorePermit, Semaphore}; -use tokio::task::{self, JoinHandle}; pub(crate) struct WalReplayOptions { - /// Limits concurrent full-object WAL prefetch. - pub(crate) prefetch: WalReplaySettings, - /// The target maximum number of bytes in each returned table. WAL replay only - /// splits between complete WAL SSTs, so a returned table may exceed this if a - /// single WAL SST is larger. + /// splits between write batches (all rows of one commit stay in one table), so + /// a returned table may exceed this if a single write batch is larger. pub(crate) max_memtable_bytes: usize, /// The minimum seq number to replay. If unset, will replay all /// entries after `last_l0_seq` in the manifest. pub(crate) min_seq: Option, - - pub(crate) source: ExactWalReplaySource, - - /// Tracks spawned fetches for a live reader. Writer opening leaves this - /// unset because its open future already owns the entire replay lifetime. - pub(crate) task_scope: Option, -} - -#[derive(Clone, Copy, Debug, Eq, PartialEq)] -pub(crate) enum ExactWalReplaySource { - /// A writable database is recovering before it can serve writes. - WriterOpen, - /// A reader is recovering its initial exact view. - ReaderOpen, - /// A running reader lost or advanced beyond its checkpoint and must recover exactly. - CheckpointRecovery, -} - -#[derive(Clone, Copy, Debug, Eq, PartialEq)] -pub(crate) enum RuntimeWalReplaySource { - /// Replay of the exact range declared by a newly applied manifest. - Manifest, - /// Recurring exact-next WAL tail while the reader remains open. - Tail, -} - -pub(crate) struct RuntimeWalReplayOptions { - /// The target maximum number of bytes in each returned table. Runtime - /// replay splits only between complete WAL SSTs. - pub(crate) max_memtable_bytes: usize, - - /// Options passed to the range-read SST iterators. - pub(crate) sst_iter_options: SstIteratorOptions, - - /// The minimum sequence number to replay. - pub(crate) min_seq: Option, - - /// Optional fairness boundary used by live tailing. Exact recovery and - /// manifest-range replay leave this unset to retain historical memtable - /// boundaries. - pub(crate) max_wals_per_batch: Option, - - pub(crate) source: RuntimeWalReplaySource, - - /// Exact owner for WAL-open and nested block-fetch tasks. - pub(crate) task_scope: Option, -} - -impl Default for RuntimeWalReplayOptions { - fn default() -> Self { - Self { - max_memtable_bytes: 64 * 1024 * 1024, - // Preserve the pre-full-object reader replay behavior: one fetch - // task eagerly reads up to 256 blocks through bounded range GETs. - sst_iter_options: SstIteratorOptions { - max_fetch_tasks: 1, - blocks_to_fetch: 256, - cache_blocks: true, - cache_metadata: false, - eager_spawn: true, - order: IterationOrder::Ascending, - prefix: None, - filter_context: None, - }, - min_seq: None, - max_wals_per_batch: None, - source: RuntimeWalReplaySource::Manifest, - task_scope: None, - } - } } impl Default for WalReplayOptions { fn default() -> Self { Self { - prefetch: WalReplaySettings::default(), max_memtable_bytes: 64 * 1024 * 1024, min_seq: None, - source: ExactWalReplaySource::WriterOpen, - task_scope: None, } } } @@ -134,3434 +40,461 @@ pub(crate) struct ReplayedMemtable { pub(crate) last_wal_id: u64, } -struct WalIdAndIter { - wal_id: u64, - iter: Box, - _permit: Option, -} - -struct RuntimeWalIdAndIter { - wal_id: u64, - iter: Box, -} - -#[derive(Debug)] -pub(crate) enum RuntimeWalReplayError { - MissingInitialObject { wal_id: u64, source: SlateDBError }, - Replay(SlateDBError), -} - -impl From for RuntimeWalReplayError { - fn from(error: SlateDBError) -> Self { - Self::Replay(error) - } -} - -struct WalObjectPlan { - wal_id: u64, - expected_size: Option, -} - -enum FetchedWal { - Full { - wal_id: u64, - bytes: Bytes, - permit: OwnedSemaphorePermit, - }, - Ranged { - wal_id: u64, - expected_size: u64, - }, -} - -impl FetchedWal { - fn wal_id(&self) -> u64 { - match self { - Self::Full { wal_id, .. } | Self::Ranged { wal_id, .. } => *wal_id, - } - } -} - -struct PendingWal { - wal_id: u64, - handle: JoinHandle>, -} - -struct WalBlocksIterator { - table_store: Arc, - wal: Option, - next_block: usize, - current: Option, - encoded_byte_limit: usize, - working_memory_limit: usize, - block_working_memory_limit: usize, - validation_retried: bool, - yielded_rows: bool, - #[cfg(test)] - block_lifetime_probe: Option>, -} - -struct CurrentWalBlock { - iterator: DataBlockIterator, - #[cfg(test)] - lifetime_probe: Option>, -} - -impl CurrentWalBlock { - fn new( - iterator: DataBlockIterator, - #[cfg(test)] lifetime_probe: Option>, - ) -> Self { - #[cfg(test)] - if let Some(probe) = &lifetime_probe { - let previous = probe - .active - .fetch_add(1, std::sync::atomic::Ordering::SeqCst); - probe - .peak - .fetch_max(previous + 1, std::sync::atomic::Ordering::SeqCst); - } - Self { - iterator, - #[cfg(test)] - lifetime_probe, - } - } -} - -#[cfg(test)] -impl Drop for CurrentWalBlock { - fn drop(&mut self) { - if let Some(probe) = &self.lifetime_probe { - let previous = probe - .active - .fetch_sub(1, std::sync::atomic::Ordering::SeqCst); - assert_eq!(previous, 1, "decoded WAL blocks overlapped"); - } - } -} - -#[cfg(test)] -#[derive(Default)] -struct BlockLifetimeProbe { - active: std::sync::atomic::AtomicUsize, - peak: std::sync::atomic::AtomicUsize, -} - -enum WalBlockSource { - Full(DecodedWalSstData), - Ranged(RangedWalSstData), -} - -impl WalBlockSource { - fn block_count(&self) -> usize { - match self { - Self::Full(wal) => wal.index.borrow().block_meta().len(), - Self::Ranged(wal) => wal.index.borrow().block_meta().len(), - } - } - - fn format_version(&self) -> u16 { - match self { - Self::Full(wal) => wal.format_version, - Self::Ranged(wal) => wal.format_version, - } - } - - fn retained_decode_bytes(&self) -> usize { - match self { - Self::Full(wal) => wal.retained_decode_bytes, - Self::Ranged(wal) => wal.retained_decode_bytes, - } - } -} - -impl WalBlocksIterator { - fn new( - table_store: Arc, - wal: DecodedWalSstData, - encoded_byte_limit: usize, - working_memory_limit: usize, - ) -> Result { - Self::from_source( - table_store, - WalBlockSource::Full(wal), - encoded_byte_limit, - working_memory_limit, - ) - } - - fn new_ranged( - table_store: Arc, - wal: RangedWalSstData, - encoded_byte_limit: usize, - working_memory_limit: usize, - ) -> Result { - Self::from_source( - table_store, - WalBlockSource::Ranged(wal), - encoded_byte_limit, - working_memory_limit, - ) - } - - fn from_source( - table_store: Arc, - wal: WalBlockSource, - encoded_byte_limit: usize, - working_memory_limit: usize, - ) -> Result { - let block_working_memory_limit = (working_memory_limit / 2).min( - working_memory_limit - .checked_sub(wal.retained_decode_bytes()) - .ok_or(SlateDBError::WalReplayMemoryLimitExceeded { - kind: "WAL metadata, index, and block", - required_bytes: wal.retained_decode_bytes(), - limit_bytes: working_memory_limit, - })?, - ); - Ok(Self { - table_store, - wal: Some(wal), - next_block: 0, - current: None, - encoded_byte_limit, - working_memory_limit, - block_working_memory_limit, - validation_retried: false, - yielded_rows: false, - #[cfg(test)] - block_lifetime_probe: None, - }) - } - - #[cfg(test)] - fn observe_block_lifetimes_with(mut self, probe: Arc) -> Self { - self.block_lifetime_probe = Some(probe); - self - } -} - -#[async_trait] -impl RowEntryIterator for WalBlocksIterator { - async fn init(&mut self) -> Result<(), SlateDBError> { - Ok(()) - } - - async fn next(&mut self) -> Result, SlateDBError> { - loop { - if let Some(current) = &mut self.current { - if let Some(entry) = current.iterator.next().await? { - self.yielded_rows = true; - return Ok(Some(entry)); - } - } - self.current.take(); - let wal = self.wal.as_ref().ok_or(SlateDBError::InvalidDBState)?; - let block_count = wal.block_count(); - if self.next_block >= block_count { - return Ok(None); - } - debug_assert!( - self.current.is_none(), - "exhausted WAL block must be released before decoding the next block" - ); - let decoded = match wal { - WalBlockSource::Full(wal) => { - self.table_store - .decode_wal_block(wal, self.next_block, self.block_working_memory_limit) - .await - } - WalBlockSource::Ranged(wal) => { - self.table_store - .read_ranged_wal_block( - wal, - self.next_block, - self.block_working_memory_limit, - ) - .await - } - }; - let block = match decoded { - Ok(block) => block, - Err(error) => { - let Some(reason) = error.maybe_validation_retry_reason() else { - return Err(error); - }; - if self.validation_retried || self.yielded_rows { - return Err(error); - } - let old_wal = self.wal.take().ok_or(SlateDBError::InvalidDBState)?; - let WalBlockSource::Full(old_wal) = old_wal else { - return Err(error); - }; - let wal_id = old_wal.wal_id; - let expected_size = old_wal.object_bytes.len(); - drop(old_wal); - self.validation_retried = true; - let replacement = self - .table_store - .refetch_wal_sst_after_validation( - wal_id, - expected_size, - self.encoded_byte_limit, - self.working_memory_limit, - reason, - ) - .await?; - let DecodedWalSst::Data(replacement) = replacement else { - return Err(SlateDBError::CorruptSst { - reason: "nonempty WAL changed into a fence during validation retry", - path: None, - }); - }; - self.block_working_memory_limit = (self.working_memory_limit / 2).min( - self.working_memory_limit - .checked_sub(replacement.retained_decode_bytes) - .ok_or(SlateDBError::WalReplayMemoryLimitExceeded { - kind: "WAL metadata, index, and block", - required_bytes: replacement.retained_decode_bytes, - limit_bytes: self.working_memory_limit, - })?, - ); - self.wal = Some(WalBlockSource::Full(*replacement)); - continue; - } - }; - self.next_block += 1; - let wal = self.wal.as_ref().ok_or(SlateDBError::InvalidDBState)?; - self.current = Some(CurrentWalBlock::new( - DataBlockIterator::new(block, wal.format_version(), IterationOrder::Ascending)?, - #[cfg(test)] - self.block_lifetime_probe.clone(), - )); - } - } - - async fn seek(&mut self, _next_key: &[u8]) -> Result<(), SlateDBError> { - Err(SlateDBError::InvalidDBState) - } -} - -struct IteratorHolder { - initialized: bool, - current_iter: Option, -} - -impl IteratorHolder { - fn new() -> Self { - Self { - initialized: false, - current_iter: None, - } - } - - fn is_finished(&self) -> bool { - self.initialized && self.current_iter.is_none() - } - - fn advance(&mut self, iterator: Option) { - self.initialized = true; - self.current_iter = iterator; - } - - fn reset(&mut self) { - self.initialized = false; - self.current_iter = None; - } -} - -/// Replays a finite WAL range through the established bounded SST range-read -/// iterator. This path deliberately performs no object-store LIST and retains -/// only a small ordered prefetch window. -pub(crate) struct RuntimeWalReplayIterator { - options: RuntimeWalReplayOptions, - wal_id_range: Range, +pub(crate) struct WalReplayIterator { + options: WalReplayOptions, table_store: Arc, - current_iter: IteratorHolder, - next_iters: VecDeque, RuntimeWalReplayError>>>, + wal_iter: Box, + terminal_result: Option>, + /// The greatest WAL ID such that it and every WAL file before it in the replay + /// range are fully applied to returned tables. Tables are tagged with this + /// conservative watermark so that a table ending mid-file never claims a WAL + /// file it only partially contains. + last_consumed_wal_file_id: u64, last_tick: i64, last_seq: u64, min_seq: u64, - next_wal_id: u64, } -const RUNTIME_WAL_PREFETCH: usize = 4; - -impl RuntimeWalReplayIterator { +impl WalReplayIterator { + #[cfg(test)] pub(crate) fn range( wal_id_range: Range, db_state: &ManifestCore, - options: RuntimeWalReplayOptions, + iterator_options: SlateDbWalIteratorOptions, + replay_options: WalReplayOptions, + table_store: Arc, + wal_store: Arc, + ) -> Result { + let wal_iter = SlateDbWalIterator::range( + wal_id_range.start, + WalIteratorEndBound::Exclusive(wal_id_range.end), + iterator_options, + wal_store, + )?; + Self::for_wal_iterator(Box::new(wal_iter), db_state, replay_options, table_store) + } + + pub(crate) fn for_wal_iterator( + wal_iter: Box, + db_state: &ManifestCore, + options: WalReplayOptions, table_store: Arc, ) -> Result { + // load the last seq number from manifest, and use it as the starting seq number to avoid + // replaying the entries that are already in the L0 SST. while replaying the WALs, we'll + // update the last seq number to the max seq number, and this final `last_seq` will be passed + // to the db_state for the further writes. let min_seq = options.min_seq.unwrap_or(db_state.last_l0_seq); let last_seq = db_state.last_l0_seq; let last_tick = db_state.last_l0_clock_tick; - let next_wal_id = wal_id_range.start; - let mut replay = Self { + + Ok(WalReplayIterator { options, - wal_id_range, table_store, - current_iter: IteratorHolder::new(), - next_iters: VecDeque::new(), + wal_iter, + terminal_result: None, + last_consumed_wal_file_id: db_state.replay_after_wal_id, last_tick, last_seq, min_seq, - next_wal_id, - }; - while replay.maybe_load_next_iter() {} - info!( - "SlateDB runtime WAL replay initialized [source={:?}, replay_start_wal_id={}, replay_end_wal_id={}, prefetch={}]", - replay.options.source, - replay.wal_id_range.start, - replay.wal_id_range.end, - RUNTIME_WAL_PREFETCH, - ); - Ok(replay) + }) } - fn maybe_load_next_iter(&mut self) -> bool { - if !self.wal_id_range.contains(&self.next_wal_id) - || self.next_iters.len() >= RUNTIME_WAL_PREFETCH - { - return false; + /// Get the next table replayed from the WAL. Replay accumulates write batches + /// until the returned table reaches [`WalReplayOptions::max_memtable_bytes`]. + /// Tables are only split between write batches — all rows sharing a commit seq + /// stay in one table, and batches are applied in ascending seq order — so a + /// returned table may exceed the target when a single write batch is larger. + /// + /// The returned table's `last_wal_id` is a conservative watermark: the greatest + /// WAL ID such that it and every WAL file before it are fully contained in the + /// tables returned so far. A table that ends mid-file is tagged with the last + /// fully replayed WAL ID, so replaying from `last_wal_id + 1` and dropping rows + /// with seq <= the table's `last_seq` never misses or duplicates a commit. + pub(crate) async fn next(&mut self) -> Result, SlateDBError> { + if let Some(terminal_result) = self.terminal_result.clone() { + return terminal_result.map(|_v| None); } - let wal_id = self.next_wal_id; - self.next_wal_id += 1; - let table_store = Arc::clone(&self.table_store); - let sst_iter_options = self.options.sst_iter_options.clone(); - let task_scope = self.options.task_scope.clone(); - let nested_scope = task_scope.clone(); - let task = async move { - let sst = match table_store.open_runtime_wal_sst(wal_id).await { - Ok(sst) => sst, - Err(RuntimeWalOpenError::Replay(SlateDBError::EmptySSTable)) => { - return Ok(Some(RuntimeWalIdAndIter { - wal_id, - iter: Box::new(EmptyIterator::new()), - })); - } - Err(RuntimeWalOpenError::MissingInitialObject(error)) => { - return Err(RuntimeWalReplayError::MissingInitialObject { - wal_id, - source: error, - }); + let table = WritableKVTable::new(); + let mut applied_any = false; + + loop { + let writes = match self.wal_iter.next().await { + Ok(Some(writes)) => writes, + Ok(None) => { + // we've reached the end of iteration, mark the iterator as done + self.terminal_result = Some(Ok(())); + break; } - Err(RuntimeWalOpenError::Replay(error)) => { - return Err(RuntimeWalReplayError::Replay(error)); + // Hold the error back so the write batches already applied to this + // table are returned first. `DbReader` treats a missing WAL file as + // the end of the WAL, so rows replayed before the error must not be + // dropped with it. + Err(err) => { + self.terminal_result = Some(Err(err.into())); + break; } }; - let iter = if let Some(scope) = nested_scope { - SstIterator::new_owned_initialized_scoped( - .., - SsTableView::identity(sst), - Arc::clone(&table_store), - sst_iter_options, - scope, - ) - .await - } else { - SstIterator::new_owned_initialized( - .., - SsTableView::identity(sst), - Arc::clone(&table_store), - sst_iter_options, - ) - .await - } - .map_err(RuntimeWalReplayError::Replay)?; - Ok(iter.map(|iter| RuntimeWalIdAndIter { - wal_id, - iter: Box::new(iter) as Box, - })) - }; - let handle = if let Some(scope) = task_scope.as_ref() { - scope.spawn(task) - } else { - task::spawn(task) - }; - self.next_iters.push_back(handle); - true - } - async fn advance_current_iter(&mut self) -> Result<(), RuntimeWalReplayError> { - let next_iter = if let Some(join_handle) = self.next_iters.pop_front() { - match join_handle.await { - Ok(result) => result?, - Err(join_error) => { - let task_name = format!("runtime_wal_replay[{:?}]", self.wal_id_range); - if let Ok(panic_error) = join_error.try_into_panic() { - error!( - "runtime WAL replay task panicked [task_name={}, panic={}]", - task_name, - panic_string(&panic_error), - ); - return Err(RuntimeWalReplayError::Replay( - SlateDBError::BackgroundTaskPanic(task_name), - )); - } - return Err(RuntimeWalReplayError::Replay( - SlateDBError::BackgroundTaskCancelled(task_name), - )); - } - } - } else { - None - }; - self.current_iter.advance(next_iter); - Ok(()) - } + applied_any = true; + assert!( + writes.last_consumed_wal_file_id >= self.last_consumed_wal_file_id, + "WAL iterator moved its consumed file watermark backwards" + ); + self.last_consumed_wal_file_id = writes.last_consumed_wal_file_id; - pub(crate) async fn next(&mut self) -> Result, RuntimeWalReplayError> { - if self.current_iter.is_finished() { - return Ok(None); - } - if !self.current_iter.initialized { - self.advance_current_iter().await?; - } + for row_entry in writes.rows { + // skip the entries that are already in the L0 SST. + if row_entry.seq <= self.min_seq { + continue; + } - let table = WritableKVTable::new(); - let mut last_wal_id = 0; - let mut replayed_wals = 0_usize; - while !self.current_iter.is_finished() { - if let Some(wal) = &mut self.current_iter.current_iter { - while let Some(row) = wal - .iter - .next() - .await - .map_err(RuntimeWalReplayError::Replay)? - { - if row.seq <= self.min_seq { - continue; - } - if let Some(timestamp) = row.create_ts { - self.last_tick = self.last_tick.max(timestamp); - } - self.last_seq = self.last_seq.max(row.seq); - table.put(row); + if let Some(ts) = row_entry.create_ts { + self.last_tick = self.last_tick.max(ts); } + self.last_seq = self.last_seq.max(row_entry.seq); + table.put(row_entry); + } - last_wal_id = wal.wal_id; - replayed_wals = replayed_wals.saturating_add(1); - let metadata = table.metadata(); - let estimated_bytes = self.table_store.estimate_encoded_size_compacted( - metadata.entry_num, - metadata.entries_size_in_bytes, - ); - if (!table.is_empty() && estimated_bytes >= self.options.max_memtable_bytes) - || self - .options - .max_wals_per_batch - .is_some_and(|limit| replayed_wals >= limit.get()) - { - self.current_iter.reset(); + if !table.is_empty() { + let meta = table.metadata(); + let estimated_bytes = self + .table_store + .estimate_encoded_size_compacted(meta.entry_num, meta.entries_size_in_bytes); + if estimated_bytes >= self.options.max_memtable_bytes { break; } } - - self.maybe_load_next_iter(); - self.advance_current_iter().await?; } - Ok((last_wal_id > 0).then_some(ReplayedMemtable { - table, - last_tick: self.last_tick, - last_seq: self.last_seq, - last_wal_id, - })) - } -} - -impl Drop for RuntimeWalReplayIterator { - fn drop(&mut self) { - for pending in &self.next_iters { - pending.abort(); + if applied_any { + // we use the applied_any check here rather than checking non-empty table size to + // ensure that if we replayed values with a lower seq num we still carry the + // wal id in an empty replayed memtable + Ok(Some(ReplayedMemtable { + table, + last_tick: self.last_tick, + last_seq: self.last_seq, + last_wal_id: self.last_consumed_wal_file_id, + })) + } else { + self.terminal_result + .clone() + .expect("applied_any false but no terminal result") + .map(|_v| None) } } } -/// Replays one exact, contiguous WAL-ID range. -/// -/// Safety invariants: -/// - every ID is fetched directly, including IDs omitted by object listing; -/// - fetched objects are consumed strictly in WAL-ID order; -/// - encoded-byte permits remain owned until the corresponding WAL is consumed; -/// - an error permanently fails the iterator and aborts every outstanding fetch; -/// - the replay cursor is lazy, so memory does not grow with a sparse ID span. -pub(crate) struct ExactWalReplayIterator { - options: WalReplayOptions, - wal_id_range: Range, - table_store: Arc, - current_iter: IteratorHolder, - next_wal_id: u64, - listed_sizes: VecDeque<(u64, u64)>, - pending_fetches: VecDeque, - byte_semaphore: Arc, - encoded_byte_limit: usize, - working_memory_limit: usize, - last_tick: i64, - last_seq: u64, - min_seq: u64, - fetched_objects: u64, - fetched_bytes: u64, - decoded_objects: u64, - ranged_objects: u64, - peak_concurrent_objects: usize, - peak_reserved_bytes: usize, - completion_logged: bool, - failed: Option, -} +#[cfg(test)] +mod tests { + use super::{SlateDbWalIteratorOptions, WalReplayIterator, WalReplayOptions}; + use crate::block_cache_policy::BlockCachePolicy; + use crate::bytes_range::BytesRange; + use crate::format::sst::SsTableFormat; + use crate::iter::{IterationOrder, RowEntryIterator}; + use crate::manifest::ManifestCore; + use crate::mem_table::WritableKVTable; + use crate::object_stores::ObjectStores; + use crate::proptest_util::{rng, sample}; + use crate::tablestore::{TableStore, TableStoreKind}; + use crate::types::RowEntry; + use crate::wal::slatedb::store::WalTableStore; + use crate::wal::{WalError, WalIterator as WalIteratorTrait, WalRows}; + use crate::{error::SlateDBError, test_utils}; + use async_trait::async_trait; + use bytes::Bytes; + use object_store::memory::InMemory; + use object_store::path::Path; + use object_store::ObjectStore; + use proptest::test_runner::TestRng; + use rand::Rng; + use std::cmp::min; + use std::collections::btree_map::Iter; + use std::collections::{BTreeMap, VecDeque}; + use std::sync::Arc; -impl ExactWalReplayIterator { - pub(crate) async fn range( - wal_id_range: Range, - db_state: &ManifestCore, - options: WalReplayOptions, - table_store: Arc, - ) -> Result { - options.prefetch.validate()?; + struct ScriptedWalIterator { + results: VecDeque, WalError>>, + } - // load the last seq number from manifest, and use it as the starting seq number to avoid - // replaying the entries that are already in the L0 SST. while replaying the WALs, we'll - // update the last seq number to the max seq number, and this final `last_seq` will be passed - // to the db_state for the further writes. - let min_seq = options.min_seq.unwrap_or(db_state.last_l0_seq); - let last_seq = db_state.last_l0_seq; - let last_tick = db_state.last_l0_clock_tick; - let listed = table_store - .list_wal_ssts_for_replay(wal_id_range.clone()) - .await?; - let listed_wal_count = listed.len(); - let mut listed_sizes = VecDeque::with_capacity(listed_wal_count); - let mut previous_listed_wal_id = None; - for metadata in listed { - let wal_id = metadata.id.unwrap_wal_id(); - if !wal_id_range.contains(&wal_id) - || previous_listed_wal_id.is_some_and(|previous| wal_id <= previous) - { - return Err(SlateDBError::InvalidDBState); - } - previous_listed_wal_id = Some(wal_id); - listed_sizes.push_back((wal_id, metadata.metadata.size)); + #[async_trait] + impl WalIteratorTrait for ScriptedWalIterator { + async fn next(&mut self) -> Result, WalError> { + self.results.pop_front().unwrap_or(Ok(None)) } - let working_memory_limit = options.prefetch.working_memory_limit(); - let encoded_byte_limit = options.prefetch.encoded_byte_limit()?; - let byte_semaphore = Arc::new(Semaphore::new(encoded_byte_limit)); - let next_wal_id = wal_id_range.start; - - let mut replay_iter = ExactWalReplayIterator { - options, - wal_id_range, - table_store: Arc::clone(&table_store), - current_iter: IteratorHolder::new(), - next_wal_id, - listed_sizes, - pending_fetches: VecDeque::new(), - byte_semaphore, - encoded_byte_limit, - working_memory_limit, - last_tick, - last_seq, - min_seq, - fetched_objects: 0, - fetched_bytes: 0, - decoded_objects: 0, - ranged_objects: 0, - peak_concurrent_objects: 0, - peak_reserved_bytes: 0, - completion_logged: false, - failed: None, - }; - - replay_iter.fill_prefetch(); - info!( - "SlateDB exact WAL replay initialized [source={:?}, replay_start_wal_id={}, replay_end_wal_id={}, replay_wal_count={}, listed_wal_count={}, missing_size_count={}, max_concurrent_objects={}, max_inflight_bytes={}]", - replay_iter.options.source, - replay_iter.wal_id_range.start, - replay_iter.wal_id_range.end, - replay_iter - .wal_id_range - .end - .saturating_sub(replay_iter.wal_id_range.start), - listed_wal_count, - replay_iter - .wal_id_range - .end - .saturating_sub(replay_iter.wal_id_range.start) - .saturating_sub(listed_wal_count as u64), - replay_iter.options.prefetch.max_concurrent_objects, - replay_iter.options.prefetch.max_inflight_bytes, - ); - - Ok(replay_iter) } - fn spawn_fetch(&self, task: F) -> JoinHandle - where - F: std::future::Future + Send + 'static, - T: Send + 'static, - { - if let Some(scope) = self.options.task_scope.as_ref() { - scope.spawn(task) - } else { - tokio::spawn(task) - } - } - - fn fill_prefetch(&mut self) { - while self.pending_fetches.len() < self.options.prefetch.max_concurrent_objects { - if self.next_wal_id >= self.wal_id_range.end { - break; - } - while self - .listed_sizes - .front() - .is_some_and(|(wal_id, _)| *wal_id < self.next_wal_id) - { - self.listed_sizes.pop_front(); - } - let plan = WalObjectPlan { - wal_id: self.next_wal_id, - expected_size: self - .listed_sizes - .front() - .filter(|(wal_id, _)| *wal_id == self.next_wal_id) - .map(|(_, size)| *size), - }; - let byte_limit = self.encoded_byte_limit; - let reserved_bytes = plan - .expected_size - .and_then(|size| usize::try_from(size).ok()) - .unwrap_or(byte_limit) - .max(1); - if reserved_bytes > byte_limit { - let wal_id = plan.wal_id; - let expected_size = plan - .expected_size - .expect("an oversized planned WAL must have a listed size"); - self.listed_sizes.pop_front(); - self.next_wal_id += 1; - let handle = self.spawn_fetch(async move { - Ok(FetchedWal::Ranged { - wal_id, - expected_size, - }) - }); - self.pending_fetches - .push_back(PendingWal { wal_id, handle }); - self.peak_concurrent_objects = - self.peak_concurrent_objects.max(self.pending_fetches.len()); - continue; - } - let permit_count = u32::try_from(reserved_bytes) - .expect("validated WAL replay byte limit must fit in u32"); - let Ok(permit) = Arc::clone(&self.byte_semaphore).try_acquire_many_owned(permit_count) - else { - break; - }; - if plan.expected_size.is_some() { - self.listed_sizes.pop_front(); - } - self.next_wal_id += 1; - let table_store = Arc::clone(&self.table_store); - let handle = self.spawn_fetch(async move { - match table_store - .read_wal_sst_bytes(plan.wal_id, plan.expected_size, byte_limit) - .await - { - Ok(bytes) => Ok(FetchedWal::Full { - wal_id: plan.wal_id, - bytes, - permit, - }), - Err(SlateDBError::WalReplayMemoryLimitExceeded { - kind: "encoded WAL object", - required_bytes, - limit_bytes, - }) if required_bytes > limit_bytes => { - drop(permit); - Ok(FetchedWal::Ranged { - wal_id: plan.wal_id, - expected_size: u64::try_from(required_bytes).map_err(|_| { - SlateDBError::WalReplayMemoryLimitExceeded { - kind: "encoded WAL object", - required_bytes, - limit_bytes, - } - })?, - }) - } - Err(error) => Err(error), - } - }); - self.pending_fetches.push_back(PendingWal { - wal_id: plan.wal_id, - handle, - }); - self.peak_concurrent_objects = - self.peak_concurrent_objects.max(self.pending_fetches.len()); - self.peak_reserved_bytes = self.peak_reserved_bytes.max( - self.encoded_byte_limit - .saturating_sub(self.byte_semaphore.available_permits()), - ); - } - } - - async fn advance_current_iter(&mut self) -> Result<(), SlateDBError> { - self.current_iter.current_iter.take(); - self.fill_prefetch(); - let next_iter = if let Some(pending) = self.pending_fetches.pop_front() { - let fetched = match pending.handle.await { - Ok(Ok(fetched)) => fetched, - Ok(Err(slate_err)) => return Err(self.fail(slate_err)), - Err(join_err) => { - let task_name = format!("wal_replay[{:?}]", self.wal_id_range); - if let Ok(panic_err) = join_err.try_into_panic() { - error!( - "wal_replay task panicked unexpectedly. [task_name={}, panic={}]", - task_name, - panic_string(&panic_err), - ); - return Err(self.fail(SlateDBError::BackgroundTaskPanic(task_name))); - } - return Err(self.fail(SlateDBError::BackgroundTaskCancelled(task_name))); - } - }; - if pending.wal_id != fetched.wal_id() { - return Err(self.fail(SlateDBError::InvalidDBState)); - } - let (wal_id, iter, permit): (_, Box, _) = match fetched - { - FetchedWal::Full { - wal_id, - bytes, - permit, - } => { - self.fetched_objects = self.fetched_objects.saturating_add(1); - self.fetched_bytes = self - .fetched_bytes - .saturating_add(u64::try_from(bytes.len()).unwrap_or(u64::MAX)); - let expected_size = bytes.len(); - let initial_decode = self - .table_store - .decode_wal_sst(wal_id, bytes, self.working_memory_limit) - .await; - let decoded = match initial_decode { - Ok(decoded) => decoded, - Err(error) => { - let Some(reason) = error.maybe_validation_retry_reason() else { - return Err(self.fail(error)); - }; - match self - .table_store - .refetch_wal_sst_after_validation( - wal_id, - expected_size, - self.encoded_byte_limit, - self.working_memory_limit, - reason, - ) - .await - { - Ok(decoded) => decoded, - Err(error) => return Err(self.fail(error)), - } - } - }; - let iter: Box = match decoded { - DecodedWalSst::Fence => Box::new(crate::iter::EmptyIterator::new()), - DecodedWalSst::Data(wal) => Box::new( - WalBlocksIterator::new( - Arc::clone(&self.table_store), - *wal, - self.encoded_byte_limit, - self.working_memory_limit, - ) - .map_err(|error| self.fail(error))?, - ), - }; - (wal_id, iter, Some(permit)) - } - FetchedWal::Ranged { - wal_id, - expected_size, - } => { - self.ranged_objects = self.ranged_objects.saturating_add(1); - let decoded = match self - .table_store - .open_ranged_wal_sst(wal_id, expected_size, self.working_memory_limit) - .await - { - Ok(decoded) => decoded, - Err(error) => return Err(self.fail(error)), - }; - let iter: Box = match decoded { - RangedWalSst::Fence => Box::new(crate::iter::EmptyIterator::new()), - RangedWalSst::Data(wal) => Box::new( - WalBlocksIterator::new_ranged( - Arc::clone(&self.table_store), - *wal, - self.encoded_byte_limit, - self.working_memory_limit, - ) - .map_err(|error| self.fail(error))?, - ), - }; - (wal_id, iter, None) - } - }; - self.decoded_objects = self.decoded_objects.saturating_add(1); - Some(WalIdAndIter { - wal_id, - iter, - _permit: permit, - }) - } else { - None - }; - self.current_iter.advance(next_iter); - self.fill_prefetch(); - Ok(()) - } - - /// Get the next table replayed from the WAL. Replay accumulates complete WAL - /// SSTs until the returned table reaches [`WalReplayOptions::max_memtable_bytes`], - /// unless it is the final table replayed from the WAL. The final table may even - /// be empty since writers use an empty WAL to fence zombie writers. The empty - /// table must still be returned so that replay logic can account for the latest - /// WAL ID. - /// - /// The returned table may exceed [`WalReplayOptions::max_memtable_bytes`] when - /// a complete WAL SST is larger than the configured target, because replay - /// must not split a WAL SST across replayed memtables. - pub(crate) async fn next(&mut self) -> Result, SlateDBError> { - if let Some(error) = &self.failed { - return Err(error.clone()); - } - if self.current_iter.is_finished() { - self.log_completion(); - return Ok(None); - } - - if !self.current_iter.initialized { - self.advance_current_iter().await?; - } - - let table = WritableKVTable::new(); - let mut last_wal_id = 0; - - while !self.current_iter.is_finished() { - if let Some(wal_id_and_iter) = &mut self.current_iter.current_iter { - loop { - let row_entry = match wal_id_and_iter.iter.next().await { - Ok(Some(row_entry)) => row_entry, - Ok(None) => break, - Err(error) => return Err(self.fail(error)), - }; - // skip the entries that are already in the L0 SST. - if row_entry.seq <= self.min_seq { - continue; - } - - if let Some(ts) = row_entry.create_ts { - self.last_tick = self.last_tick.max(ts); - } - self.last_seq = self.last_seq.max(row_entry.seq); - table.put(row_entry); - } - - last_wal_id = wal_id_and_iter.wal_id; - let replayed_wal_count = last_wal_id - .saturating_sub(self.wal_id_range.start) - .saturating_add(1); - if replayed_wal_count.is_multiple_of(128) { - let metadata = table.metadata(); - info!( - "SlateDB WAL replay progress [replay_start_wal_id={}, replay_end_wal_id={}, replay_wal_count={}, last_replayed_wal_id={}, replayed_wal_count={}, replayed_entries={}, replayed_bytes={}]", - self.wal_id_range.start, - self.wal_id_range.end, - self.wal_id_range - .end - .saturating_sub(self.wal_id_range.start), - last_wal_id, - replayed_wal_count, - metadata.entry_num, - metadata.entries_size_in_bytes - ); - } - - let meta = table.metadata(); - let estimated_bytes = self - .table_store - .estimate_encoded_size_compacted(meta.entry_num, meta.entries_size_in_bytes); - if !table.is_empty() && estimated_bytes >= self.options.max_memtable_bytes { - self.current_iter.reset(); - break; - } - } - - self.advance_current_iter().await? - } - - if last_wal_id > 0 { - Ok(Some(ReplayedMemtable { - table, - last_tick: self.last_tick, - last_seq: self.last_seq, - last_wal_id, - })) - } else { - self.log_completion(); - Ok(None) - } - } - - fn log_completion(&mut self) { - if self.completion_logged { - return; - } - self.completion_logged = true; - info!( - "SlateDB exact WAL replay completed [source={:?}, replay_start_wal_id={}, replay_end_wal_id={}, fetched_objects={}, fetched_bytes={}, ranged_objects={}, decoded_objects={}, peak_concurrent_objects={}, peak_reserved_bytes={}]", - self.options.source, - self.wal_id_range.start, - self.wal_id_range.end, - self.fetched_objects, - self.fetched_bytes, - self.ranged_objects, - self.decoded_objects, - self.peak_concurrent_objects, - self.peak_reserved_bytes, - ); - } - - fn fail(&mut self, error: SlateDBError) -> SlateDBError { - for pending in self.pending_fetches.drain(..) { - pending.handle.abort(); - } - self.current_iter.current_iter.take(); - self.failed = Some(error.clone()); - error - } -} - -impl Drop for ExactWalReplayIterator { - fn drop(&mut self) { - for pending in &self.pending_fetches { - pending.handle.abort(); - } - } -} - -#[cfg(test)] -mod tests { - use super::{ - ExactWalReplayIterator, RuntimeWalReplayError, RuntimeWalReplayIterator, - RuntimeWalReplayOptions, WalReplayOptions, - }; - use crate::block_cache_policy::BlockCachePolicy; - use crate::bytes_range::BytesRange; - use crate::config::WalReplaySettings; - use crate::db_state::SsTableId; - use crate::format::sst::{BlockTransformer, SsTableFormat}; - use crate::iter::{IterationOrder, RowEntryIterator}; - use crate::manifest::ManifestCore; - use crate::mem_table::WritableKVTable; - use crate::object_stores::ObjectStores; - use crate::proptest_util::{rng, sample}; - use crate::tablestore::{DecodedWalSst, TableStore, TableStoreKind}; - use crate::test_utils::{FlakyObjectStore, GatedObjectStore, RecordingObjectStore}; - use crate::types::{RowEntry, ValueDeletable}; - use crate::{error::SlateDBError, test_utils}; - use async_trait::async_trait; - use bytes::Bytes; - use futures::stream::BoxStream; - use futures::FutureExt; - use object_store::memory::InMemory; - use object_store::path::Path; - use object_store::{ - CopyOptions, GetOptions, GetResult, ListResult, MultipartUpload, ObjectMeta, ObjectStore, - ObjectStoreExt, PutMultipartOptions, PutOptions, PutPayload, PutResult, RenameOptions, - }; - use proptest::prelude::*; - use proptest::test_runner::TestRng; - use rand::Rng; - use std::cmp::min; - use std::collections::btree_map::Iter; - use std::collections::BTreeMap; - use std::sync::atomic::{AtomicBool, Ordering}; - use std::sync::Arc; - use std::time::Duration; - use tokio::sync::Notify; - - struct HistoricalXorTransformer; - - #[async_trait] - impl BlockTransformer for HistoricalXorTransformer { - fn max_decoded_len(&self, encoded_len: usize) -> Option { - Some(encoded_len) - } - - async fn encode(&self, data: Bytes) -> Result { - Ok(Bytes::from( - data.iter().map(|byte| byte ^ 0xa5).collect::>(), - )) - } - - async fn decode(&self, data: Bytes) -> Result { - self.encode(data).await - } - } - - struct UnboundedTransformer { - decode_called: Arc, - } - - #[async_trait] - impl BlockTransformer for UnboundedTransformer { - fn max_decoded_len(&self, _encoded_len: usize) -> Option { - None - } - - async fn encode(&self, data: Bytes) -> Result { - Ok(data) - } - - async fn decode(&self, data: Bytes) -> Result { - self.decode_called.store(true, Ordering::SeqCst); - Ok(data) - } - } - - #[cfg(test)] - use crate::sst_builder::BlockFormat; - - #[derive(Debug)] - struct PausedFirstWalGetStore { - inner: Arc, - first_path: Path, - panic_path: Option, - first_started: Arc, - later_get_returned: Arc, - first_cancelled: Arc, - release_first: Arc, - } - - impl PausedFirstWalGetStore { - fn new(inner: Arc, first_path: Path) -> Self { - Self { - inner, - first_path, - panic_path: None, - first_started: Arc::new(AtomicBool::new(false)), - later_get_returned: Arc::new(AtomicBool::new(false)), - first_cancelled: Arc::new(AtomicBool::new(false)), - release_first: Arc::new(Notify::new()), - } - } - - fn panic_on(mut self, path: Path) -> Self { - self.panic_path = Some(path); - self - } - - async fn wait_until(flag: &AtomicBool) { - while !flag.load(Ordering::Acquire) { - tokio::task::yield_now().await; - } - } - } - - impl std::fmt::Display for PausedFirstWalGetStore { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - write!(f, "PausedFirstWalGetStore({})", self.inner) - } - } - - struct PendingGetGuard { - cancelled: Arc, - completed: bool, - } - - impl Drop for PendingGetGuard { - fn drop(&mut self) { - if !self.completed { - self.cancelled.store(true, Ordering::Release); - } - } - } - - #[async_trait] - impl ObjectStore for PausedFirstWalGetStore { - async fn get_opts( - &self, - location: &Path, - options: GetOptions, - ) -> object_store::Result { - if !options.head && self.panic_path.as_ref() == Some(location) { - panic!("injected WAL fetch panic for {location}"); - } - if !options.head && location == &self.first_path { - let mut guard = PendingGetGuard { - cancelled: Arc::clone(&self.first_cancelled), - completed: false, - }; - self.first_started.store(true, Ordering::Release); - self.release_first.notified().await; - let result = self.inner.get_opts(location, options).await; - guard.completed = true; - return result; - } - let result = self.inner.get_opts(location, options).await; - if location != &self.first_path { - self.later_get_returned.store(true, Ordering::Release); - } - result - } - - async fn put_opts( - &self, - location: &Path, - payload: PutPayload, - options: PutOptions, - ) -> object_store::Result { - self.inner.put_opts(location, payload, options).await - } - - async fn put_multipart_opts( - &self, - location: &Path, - options: PutMultipartOptions, - ) -> object_store::Result> { - self.inner.put_multipart_opts(location, options).await - } - - fn delete_stream( - &self, - locations: BoxStream<'static, object_store::Result>, - ) -> BoxStream<'static, object_store::Result> { - self.inner.delete_stream(locations) - } - - fn list( - &self, - prefix: Option<&Path>, - ) -> BoxStream<'static, object_store::Result> { - self.inner.list(prefix) - } - - fn list_with_offset( - &self, - prefix: Option<&Path>, - offset: &Path, - ) -> BoxStream<'static, object_store::Result> { - self.inner.list_with_offset(prefix, offset) - } - - async fn list_with_delimiter( - &self, - prefix: Option<&Path>, - ) -> object_store::Result { - self.inner.list_with_delimiter(prefix).await - } - - async fn copy_opts( - &self, - from: &Path, - to: &Path, - options: CopyOptions, - ) -> object_store::Result<()> { - self.inner.copy_opts(from, to, options).await - } - - async fn rename_opts( - &self, - from: &Path, - to: &Path, - options: RenameOptions, - ) -> object_store::Result<()> { - self.inner.rename_opts(from, to, options).await - } - } - - impl ExactWalReplayIterator { + impl WalReplayIterator { async fn all_wal_ids( db_state: &ManifestCore, options: WalReplayOptions, table_store: Arc, + wal_store: Arc, ) -> Result { let wal_id_start = db_state.replay_after_wal_id + 1; - let wal_id_end = table_store + let wal_id_end = wal_store .last_seen_wal_id(db_state.replay_after_wal_id) .await?; let wal_id_range = wal_id_start..(wal_id_end + 1); - Self::range(wal_id_range, db_state, options, table_store).await - } - } - - #[tokio::test] - async fn runtime_replay_uses_range_reads_without_listing() { - let recording = Arc::new(RecordingObjectStore::new(Arc::new(InMemory::new()))); - let object_store: Arc = recording.clone(); - let table_store = test_table_store_with_object_store(object_store); - write_empty_wal(1, Arc::clone(&table_store)).await.unwrap(); - - let mut runtime = RuntimeWalReplayIterator::range( - 1..2, - &ManifestCore::new(), - RuntimeWalReplayOptions::default(), - Arc::clone(&table_store), - ) - .unwrap(); - assert!(runtime.next().await.unwrap().is_some()); - assert!(runtime.next().await.unwrap().is_none()); - assert_eq!(recording.list_calls(), 0); - assert!(!recording.get_range_sizes().is_empty()); - - let mut exact = ExactWalReplayIterator::range( - 1..2, - &ManifestCore::new(), - WalReplayOptions::default(), - table_store, - ) - .await - .unwrap(); - assert!(exact.next().await.unwrap().is_some()); - assert!(recording.list_calls() > 0); - } - - #[tokio::test] - async fn runtime_replay_treats_metadata_not_found_as_corruption_not_caught_up() { - let inner: Arc = Arc::new(InMemory::new()); - let gated = Arc::new(GatedObjectStore::new(inner)); - let object_store: Arc = gated.clone(); - let table_store = test_table_store_with_object_store(object_store); - let mut rng = rng::new_test_rng(None); - let entries = sample::table(&mut rng, 1, 16); - let mut entries = entries.iter(); - write_wal(1, 1, &mut entries, 1, Arc::clone(&table_store)) - .await - .unwrap(); - - // The initial HEAD and footer read establish that the exact object - // exists. A later metadata 404 must remain a replay error. - let prior_range_arrivals = gated.get_opts_gate.arrivals(); - gated.get_opts_gate.close(); - gated.get_opts_gate.admit(1); - let mut runtime = RuntimeWalReplayIterator::range( - 1..2, - &ManifestCore::new(), - RuntimeWalReplayOptions::default(), - table_store, - ) - .unwrap(); - let replay = tokio::spawn(async move { runtime.next().await }); - tokio::time::timeout( - Duration::from_secs(2), - gated - .get_opts_gate - .wait_for_arrivals(prior_range_arrivals + 2), - ) - .await - .unwrap_or_else(|_| { - panic!( - "only {} range reads arrived", - gated.get_opts_gate.arrivals() - ) - }); - gated - .get_opts_gate - .set_error(|| object_store::Error::NotFound { - path: "injected-metadata-404".to_string(), - source: Box::new(std::io::Error::other("injected metadata 404")), - }); - gated.get_opts_gate.release(); - - let replay = tokio::time::timeout(Duration::from_secs(2), replay) - .await - .unwrap_or_else(|_| { - panic!( - "runtime replay remained blocked after {} range reads", - gated.get_opts_gate.arrivals() - ) - }) - .unwrap(); - let error = match replay { - Ok(_) => panic!("metadata 404 unexpectedly replayed"), - Err(error) => error, - }; - let RuntimeWalReplayError::Replay(error) = error else { - panic!("metadata 404 was incorrectly classified as caught up"); - }; - assert!(error.has_object_store_not_found()); - } - - #[tokio::test] - async fn runtime_replay_returns_corruption_for_a_truncated_footer_without_panicking() { - let inner: Arc = Arc::new(InMemory::new()); - let flaky = Arc::new(FlakyObjectStore::new(inner, 0).with_truncate_get_range_bytes(1, 2)); - let object_store: Arc = flaky; - let table_store = test_table_store_with_object_store(object_store); - let entries = BTreeMap::from([(Bytes::from_static(b"key"), Bytes::from_static(b"value"))]); - let mut entries = entries.iter(); - write_wal(1, 1, &mut entries, 1, Arc::clone(&table_store)) - .await - .unwrap(); - let mut runtime = RuntimeWalReplayIterator::range( - 1..2, - &ManifestCore::new(), - RuntimeWalReplayOptions::default(), - table_store, - ) - .unwrap(); - - let replay = std::panic::AssertUnwindSafe(runtime.next()) - .catch_unwind() - .await; - let result = replay.expect("malformed object-store bytes must never panic"); - let error = match result { - Ok(_) => panic!("truncated footer unexpectedly replayed"), - Err(error) => error, - }; - let RuntimeWalReplayError::Replay(SlateDBError::CorruptSst { .. }) = error else { - panic!("truncated footer did not return typed corruption"); - }; - } - - #[tokio::test] - async fn should_replay_empty_wal() { - let table_store = test_table_store(); - write_empty_wal(1, Arc::clone(&table_store)).await.unwrap(); - let mut replay_iter = ExactWalReplayIterator::all_wal_ids( - &ManifestCore::new(), - WalReplayOptions::default(), - Arc::clone(&table_store), - ) - .await - .unwrap(); - - let Some(table) = replay_iter.next().await.unwrap() else { - panic!("Expected empty table to be returned from iterator") - }; - - assert_eq!(table.last_wal_id, 1); - assert_eq!(table.last_seq, 0); - assert!(table.table.is_empty()); - assert_eq!(table.last_tick, i64::MIN); - assert!(replay_iter.next().await.unwrap().is_none()); - } - - #[tokio::test] - async fn should_replay_zero_byte_wal_fence() { - let table_store = test_table_store(); - table_store.write_wal_fence(1).await.unwrap(); - let mut replay_iter = ExactWalReplayIterator::all_wal_ids( - &ManifestCore::new(), - WalReplayOptions::default(), - Arc::clone(&table_store), - ) - .await - .unwrap(); - - let Some(table) = replay_iter.next().await.unwrap() else { - panic!("Expected empty table to be returned from iterator") - }; - - assert_eq!(table.last_wal_id, 1); - assert_eq!(table.last_seq, 0); - assert!(table.table.is_empty()); - assert_eq!(table.last_tick, i64::MIN); - assert!(replay_iter.next().await.unwrap().is_none()); - } - - #[tokio::test] - async fn should_replay_zero_byte_wal_fence_before_real_wal() { - let table_store = test_table_store(); - table_store.write_wal_fence(1).await.unwrap(); - - let row = RowEntry::new_value(b"key", b"value", 1); - let mut builder = table_store.wal_table_builder(); - builder.add(row.clone()).await.unwrap(); - let encoded_sst = builder.build().await.unwrap(); - table_store - .write_sst(&SsTableId::Wal(2), &encoded_sst) - .await - .unwrap(); - - let mut replay_iter = ExactWalReplayIterator::all_wal_ids( - &ManifestCore::new(), - WalReplayOptions::default(), - Arc::clone(&table_store), - ) - .await - .unwrap(); - - let Some(replayed_table) = replay_iter.next().await.unwrap() else { - panic!("Expected table to be returned from iterator") - }; - assert_eq!(replayed_table.last_wal_id, 2); - assert_eq!(replayed_table.last_seq, 1); - - let mut iter = replayed_table.table.table().iter(); - test_utils::assert_iterator(&mut iter, vec![row]).await; - assert!(replay_iter.next().await.unwrap().is_none()); - } - - #[tokio::test] - async fn should_replay_zero_byte_wal_fences_between_and_after_data_wals() { - let table_store = test_table_store(); - let first = RowEntry::new_value(b"first", b"value-1", 1); - let second = RowEntry::new_value(b"second", b"value-2", 2); - - for (wal_id, row) in [(1, first.clone()), (3, second.clone())] { - let mut builder = table_store.wal_table_builder(); - builder.add(row).await.unwrap(); - let encoded = builder.build().await.unwrap(); - table_store - .write_sst(&SsTableId::Wal(wal_id), &encoded) - .await - .unwrap(); - } - table_store.write_wal_fence(2).await.unwrap(); - table_store.write_wal_fence(4).await.unwrap(); - - let mut replay_iter = ExactWalReplayIterator::range( - 1..5, - &ManifestCore::new(), - WalReplayOptions::default(), - Arc::clone(&table_store), - ) - .await - .unwrap(); - - let Some(replayed) = replay_iter.next().await.unwrap() else { - panic!("expected replayed WALs"); - }; - assert_eq!(replayed.last_wal_id, 4); - assert_eq!(replayed.last_seq, 2); - let mut iter = replayed.table.table().iter(); - test_utils::assert_iterator(&mut iter, vec![first, second]).await; - assert!(replay_iter.next().await.unwrap().is_none()); - } - - #[tokio::test] - async fn should_replay_legacy_v1_wal_encoding() { - let format = SsTableFormat { - block_format: Some(BlockFormat::V1), - ..SsTableFormat::default() - }; - let table_store = test_table_store_with_format(format); - let rows = vec![ - RowEntry::new_value(b"first", b"value-1", 1), - RowEntry::new_value(b"second", b"value-2", 2), - ]; - let mut builder = table_store.table_builder(); - for row in &rows { - builder.add(row.clone()).await.unwrap(); - } - let encoded = builder.build().await.unwrap(); - table_store - .write_sst(&SsTableId::Wal(1), &encoded) - .await - .unwrap(); - - let mut replay_iter = ExactWalReplayIterator::range( - 1..2, - &ManifestCore::new(), - WalReplayOptions::default(), - Arc::clone(&table_store), - ) - .await - .unwrap(); - let Some(replayed) = replay_iter.next().await.unwrap() else { - panic!("expected legacy WAL to replay"); - }; - let mut iter = replayed.table.table().iter(); - test_utils::assert_iterator(&mut iter, rows).await; - assert_eq!(replayed.last_wal_id, 1); - assert_eq!(replayed.last_seq, 2); - } - - #[tokio::test] - async fn historical_wal_goldens_match_range_and_full_object_decoders() { - for (name, bytes) in historical_wal_fixtures() { - let mut format = SsTableFormat::default(); - if name.ends_with("-xor") { - format.block_transformer = Some(Arc::new(HistoricalXorTransformer)); - } - let store: Arc = Arc::new(InMemory::new()); - let table_store = - test_table_store_with_format_and_object_store(format, Arc::clone(&store)); - let path = Path::from("/tmp/test_kv_store/wal/00000000000000000001.sst"); - store.put(&path, bytes.clone().into()).await.unwrap(); - - let handle = table_store.open_sst(&SsTableId::Wal(1)).await.unwrap(); - let index = table_store.read_index(&handle, false).await.unwrap(); - let block_count = index.borrow().block_meta().len(); - let range_blocks = table_store - .read_blocks_using_index(&handle, index, 0..block_count, false) - .await - .unwrap(); - let mut range_rows = Vec::new(); - for block in range_blocks { - let mut iterator = crate::block_iterator::DataBlockIterator::new( - block, - handle.format_version, - IterationOrder::Ascending, - ) - .unwrap(); - while let Some(row) = iterator.next().await.unwrap() { - range_rows.push(row); - } - } - - let decoded = table_store - .decode_wal_sst(1, bytes, 64 * 1024 * 1024) - .await - .unwrap(); - let DecodedWalSst::Data(wal) = decoded else { - panic!("historical fixture {name} decoded as a fence"); - }; - let mut iterator = super::WalBlocksIterator::new( - Arc::clone(&table_store), - *wal, - 192 * 1024 * 1024, - 64 * 1024 * 1024, - ) - .unwrap(); - let mut full_rows = Vec::new(); - while let Some(row) = iterator.next().await.unwrap() { - full_rows.push(row); - } - - let ranged = table_store - .open_ranged_wal_sst(1, store.head(&path).await.unwrap().size, 64 * 1024 * 1024) - .await - .unwrap(); - let super::RangedWalSst::Data(wal) = ranged else { - panic!("historical fixture {name} range-decoded as a fence"); - }; - let mut iterator = super::WalBlocksIterator::new_ranged( - Arc::clone(&table_store), - *wal, - 192 * 1024 * 1024, - 64 * 1024 * 1024, - ) - .unwrap(); - let mut bounded_range_rows = Vec::new(); - while let Some(row) = iterator.next().await.unwrap() { - bounded_range_rows.push(row); - } - - assert_eq!(full_rows, range_rows, "fixture {name}"); - assert_eq!(bounded_range_rows, range_rows, "fixture {name}"); - assert_historical_rows(name, &full_rows); - - let expected_filtered = range_rows - .iter() - .filter(|row| row.seq > 8) - .cloned() - .collect::>(); - let mut replay = ExactWalReplayIterator::range( - 1..2, - &ManifestCore::new(), - WalReplayOptions { - min_seq: Some(8), - ..WalReplayOptions::default() - }, - Arc::clone(&table_store), - ) - .await - .unwrap(); - let replayed = replay.next().await.unwrap().unwrap(); - let mut replayed_rows = Vec::new(); - let mut replayed_iter = replayed.table.table().iter(); - while let Some(row) = replayed_iter.next().await.unwrap() { - replayed_rows.push(row); - } - assert_eq!(replayed_rows, expected_filtered, "filtered fixture {name}"); - assert!(replay.next().await.unwrap().is_none()); - - let mut runtime_replay = RuntimeWalReplayIterator::range( - 1..2, - &ManifestCore::new(), - RuntimeWalReplayOptions { - min_seq: Some(8), - ..RuntimeWalReplayOptions::default() - }, - Arc::clone(&table_store), - ) - .unwrap(); - let runtime_table = runtime_replay.next().await.unwrap().unwrap(); - let mut runtime_rows = Vec::new(); - let mut runtime_iter = runtime_table.table.table().iter(); - while let Some(row) = runtime_iter.next().await.unwrap() { - runtime_rows.push(row); - } - assert_eq!(runtime_rows, expected_filtered, "runtime fixture {name}"); - assert!(runtime_replay.next().await.unwrap().is_none()); - } - } - - #[tokio::test] - async fn historical_wal_goldens_report_identical_checksum_errors() { - for (name, bytes) in historical_wal_fixtures() { - let mut format = SsTableFormat::default(); - if name.ends_with("-xor") { - format.block_transformer = Some(Arc::new(HistoricalXorTransformer)); - } - let store: Arc = Arc::new(InMemory::new()); - let table_store = - test_table_store_with_format_and_object_store(format, Arc::clone(&store)); - let path = Path::from("/tmp/test_kv_store/wal/00000000000000000001.sst"); - let mut corrupted = bytes.to_vec(); - corrupted[0] ^= 1; - let corrupted = Bytes::from(corrupted); - store.put(&path, corrupted.clone().into()).await.unwrap(); - - let handle = table_store.open_sst(&SsTableId::Wal(1)).await.unwrap(); - let index = table_store.read_index(&handle, false).await.unwrap(); - let range_error = table_store - .read_blocks_using_index( - &handle, - index.clone(), - 0..index.borrow().block_meta().len(), - false, - ) - .await; - let Err(range_error) = range_error else { - panic!("corrupted historical fixture {name} passed range decoding"); - }; - - let decoded = table_store - .decode_wal_sst(1, corrupted, 64 * 1024 * 1024) - .await - .unwrap(); - let DecodedWalSst::Data(wal) = decoded else { - panic!("corrupted historical fixture {name} decoded as a fence"); - }; - let mut full_iterator = super::WalBlocksIterator::new( - Arc::clone(&table_store), - *wal, - 192 * 1024 * 1024, - 64 * 1024 * 1024, - ) - .unwrap(); - let full_error = full_iterator.next().await.unwrap_err(); - - let mut runtime_replay = RuntimeWalReplayIterator::range( - 1..2, - &ManifestCore::new(), - RuntimeWalReplayOptions::default(), - Arc::clone(&table_store), - ) - .unwrap(); - let runtime_error = match runtime_replay.next().await { - Ok(_) => panic!("corrupted historical fixture {name} passed runtime replay"), - Err(error) => error, - }; - let RuntimeWalReplayError::Replay(runtime_error) = runtime_error else { - panic!("corrupted historical fixture {name} looked missing"); - }; - - assert_eq!( - full_error.maybe_validation_retry_reason(), - range_error.maybe_validation_retry_reason(), - "fixture {name}" - ); - assert_eq!( - full_error.maybe_validation_retry_reason(), - Some(crate::error::RetryReason::CrcMismatch), - "fixture {name}" - ); - assert_eq!( - runtime_error.maybe_validation_retry_reason(), - full_error.maybe_validation_retry_reason(), - "runtime fixture {name}" - ); - } - } - - #[cfg(feature = "snappy")] - #[tokio::test] - async fn should_replay_compressed_wal() { - let format = SsTableFormat { - compression_codec: Some(crate::config::CompressionCodec::Snappy), - ..SsTableFormat::default() - }; - let table_store = test_table_store_with_format(format); - let row = RowEntry::new_value(b"key", &[b'x'; 16 * 1024], 1); - let mut builder = table_store.wal_table_builder(); - builder.add(row.clone()).await.unwrap(); - let encoded = builder.build().await.unwrap(); - table_store - .write_sst(&SsTableId::Wal(1), &encoded) - .await - .unwrap(); - - let mut replay_iter = ExactWalReplayIterator::range( - 1..2, - &ManifestCore::new(), - WalReplayOptions::default(), - Arc::clone(&table_store), - ) - .await - .unwrap(); - let Some(replayed) = replay_iter.next().await.unwrap() else { - panic!("expected compressed WAL to replay"); - }; - let mut iter = replayed.table.table().iter(); - test_utils::assert_iterator(&mut iter, vec![row]).await; - } - - #[tokio::test] - async fn should_finish_empty_replay_range_without_object_io() { - let recording = Arc::new(RecordingObjectStore::new(Arc::new(InMemory::new()))); - let table_store = test_table_store_with_object_store(recording.clone()); - let mut replay_iter = ExactWalReplayIterator::range( - 7..7, - &ManifestCore::new(), - WalReplayOptions::default(), - table_store, - ) - .await - .unwrap(); - - assert!(replay_iter.next().await.unwrap().is_none()); - assert!(recording.get_kinds(false).is_empty()); - assert!(recording.get_kinds(true).is_empty()); - } - - #[tokio::test] - async fn should_replay_all_entries() { - let table_store = test_table_store(); - let mut rng = rng::new_test_rng(None); - let entries = sample::table(&mut rng, 1000, 10); - let next_wal_id = write_wals(&entries, 1, &mut rng, 200, Arc::clone(&table_store)) - .await - .unwrap(); - - let mut replay_iter = ExactWalReplayIterator::all_wal_ids( - &ManifestCore::new(), - WalReplayOptions::default(), - Arc::clone(&table_store), - ) - .await - .unwrap(); - - let Some(replayed_table) = replay_iter.next().await.unwrap() else { - panic!("Expected table to be returned from iterator") - }; - assert_eq!(replayed_table.last_wal_id + 1, next_wal_id); - - let mut imm_table_iter = replayed_table.table.table().iter(); - test_utils::assert_ranged_kv_scan( - &entries, - &BytesRange::from(..), - IterationOrder::Ascending, - &mut imm_table_iter, - ) - .await; - assert!(replay_iter.next().await.unwrap().is_none()); - } - - #[tokio::test] - async fn should_issue_one_full_get_per_wal() { - let inner: Arc = Arc::new(InMemory::new()); - let recording = Arc::new(RecordingObjectStore::new(inner)); - let table_store = test_table_store_with_object_store(recording.clone()); - for wal_id in 1..=3 { - let mut builder = table_store.wal_table_builder(); - builder - .add(RowEntry::new_value( - format!("key-{wal_id}").as_bytes(), - b"value", - wal_id, - )) - .await - .unwrap(); - let encoded = builder.build().await.unwrap(); - table_store - .write_sst(&SsTableId::Wal(wal_id), &encoded) - .await - .unwrap(); - } - recording.clear(); - - let mut replay_iter = ExactWalReplayIterator::range( - 1..4, - &ManifestCore::new(), - WalReplayOptions::default(), - table_store, - ) - .await - .unwrap(); - while replay_iter.next().await.unwrap().is_some() {} - - assert_eq!(recording.get_kinds(false).len(), 3); - assert!(recording.get_kinds(true).is_empty()); - assert_eq!( - recording.get_sst_types(false), - vec![Some(crate::db_state::SstType::Wal); 3] - ); - assert_eq!(recording.get_retries(false), vec![None; 3]); - } - - #[tokio::test] - async fn should_fetch_replay_objects_only_from_dedicated_wal_store() { - let main = Arc::new(RecordingObjectStore::new(Arc::new(InMemory::new()))); - let wal = Arc::new(RecordingObjectStore::new(Arc::new(InMemory::new()))); - let table_store = Arc::new(TableStore::new( - ObjectStores::new(main.clone(), Some(wal.clone())), - SsTableFormat::default(), - Path::from("/tmp/test_kv_store"), - None, - TableStoreKind::Main, - BlockCachePolicy::default(), - )); - let mut builder = table_store.wal_table_builder(); - builder - .add(RowEntry::new_value(b"key", b"value", 1)) - .await - .unwrap(); - let encoded = builder.build().await.unwrap(); - table_store - .write_sst(&SsTableId::Wal(1), &encoded) - .await - .unwrap(); - main.clear(); - wal.clear(); - - let mut replay_iter = ExactWalReplayIterator::range( - 1..2, - &ManifestCore::new(), - WalReplayOptions::default(), - table_store, - ) - .await - .unwrap(); - while replay_iter.next().await.unwrap().is_some() {} - - assert!(main.get_kinds(false).is_empty()); - assert!(main.get_kinds(true).is_empty()); - assert_eq!(wal.get_kinds(false), vec![Some(TableStoreKind::Main)]); - assert!(wal.get_kinds(true).is_empty()); - } - - #[tokio::test] - async fn should_refetch_the_full_wal_once_after_checksum_failure() { - let inner = Arc::new(InMemory::new()); - let recording = Arc::new(RecordingObjectStore::new(inner.clone())); - let table_store = test_table_store_with_object_store(recording.clone()); - let mut builder = table_store.wal_table_builder(); - builder - .add(RowEntry::new_value(b"key", b"value", 1)) - .await - .unwrap(); - let encoded = builder.build().await.unwrap(); - table_store - .write_sst(&SsTableId::Wal(1), &encoded) - .await - .unwrap(); - let metadata = table_store - .list_wal_ssts_for_replay(1..2) - .await - .unwrap() - .pop() - .unwrap(); - let mut corrupted = inner - .get(&metadata.metadata.location) - .await - .unwrap() - .bytes() - .await - .unwrap() - .to_vec(); - corrupted[0] ^= 1; - inner - .put(&metadata.metadata.location, corrupted.into()) - .await - .unwrap(); - recording.clear(); - - let mut replay_iter = ExactWalReplayIterator::range( - 1..2, - &ManifestCore::new(), - WalReplayOptions::default(), - table_store, - ) - .await - .unwrap(); - assert!(matches!( - replay_iter.next().await, - Err(SlateDBError::ChecksumMismatch { .. }) - )); - assert_eq!(recording.get_kinds(false).len(), 2); - assert_eq!(recording.get_retries(false)[0], None); - assert!(recording.get_retries(false)[1].is_some()); - } - - #[tokio::test] - async fn should_recover_when_validation_refetch_returns_valid_wal() { - let inner = Arc::new(InMemory::new()); - let recording = Arc::new(RecordingObjectStore::new(inner)); - let table_store = test_table_store_with_object_store(recording.clone()); - let mut builder = table_store.wal_table_builder(); - builder - .add(RowEntry::new_value(b"key", b"value", 1)) - .await - .unwrap(); - let encoded = builder.build().await.unwrap(); - let valid = encoded.remaining_as_bytes(); - table_store - .write_sst(&SsTableId::Wal(1), &encoded) - .await - .unwrap(); - recording.clear(); - - let mut corrupted = valid.to_vec(); - corrupted[0] ^= 1; - let initially_decoded = table_store - .decode_wal_sst(1, Bytes::from(corrupted), 64 * 1024 * 1024) - .await - .unwrap(); - let DecodedWalSst::Data(initially_decoded) = initially_decoded else { - panic!("nonempty WAL decoded as a fence"); - }; - let Err(initial_error) = table_store - .decode_wal_block(&initially_decoded, 0, 64 * 1024 * 1024) - .await - else { - panic!("corrupt initial WAL block unexpectedly decoded"); - }; - drop(initially_decoded); - let reason = initial_error - .maybe_validation_retry_reason() - .expect("corruption must be classified for a validation retry"); - let decoded = table_store - .refetch_wal_sst_after_validation( - 1, - valid.len(), - 192 * 1024 * 1024, - 64 * 1024 * 1024, - reason, - ) - .await - .unwrap(); - - assert!(matches!(decoded, DecodedWalSst::Data(_))); - assert_eq!(recording.get_kinds(false).len(), 1); - assert!(recording.get_retries(false)[0].is_some()); - } - - #[tokio::test] - async fn should_fail_without_panicking_for_every_truncated_wal_prefix() { - let inner = Arc::new(InMemory::new()); - let table_store = test_table_store_with_object_store(inner.clone()); - let mut builder = table_store.wal_table_builder(); - for seq in 1..=8 { - builder - .add(RowEntry::new_value( - format!("key-{seq}").as_bytes(), - &[b'x'; 128], - seq, - )) - .await - .unwrap(); - } - let encoded = builder.build().await.unwrap(); - let valid = encoded.remaining_as_bytes(); - let location = Path::from("/tmp/test_kv_store/wal/00000000000000000001.sst"); - - for prefix_len in 1..valid.len() { - let truncated = valid.slice(..prefix_len); - inner - .put(&location, truncated.clone().into()) - .await - .unwrap(); - let result = std::panic::AssertUnwindSafe(table_store.decode_wal_sst( - 1, - truncated, - 64 * 1024 * 1024, - )) - .catch_unwind() - .await; - let decode = result - .unwrap_or_else(|_| panic!("WAL decoder panicked for prefix length {prefix_len}")); - assert!( - decode.is_err(), - "truncated WAL prefix {prefix_len} decoded successfully" - ); - } - } - - #[tokio::test] - async fn should_not_skip_a_missing_wal_id() { - let table_store = test_table_store(); - for wal_id in [1, 3] { - let mut builder = table_store.wal_table_builder(); - builder - .add(RowEntry::new_value( - format!("key-{wal_id}").as_bytes(), - b"value", - wal_id, - )) - .await - .unwrap(); - let encoded = builder.build().await.unwrap(); - table_store - .write_sst(&SsTableId::Wal(wal_id), &encoded) - .await - .unwrap(); - } - - let mut replay_iter = ExactWalReplayIterator::range( - 1..4, - &ManifestCore::new(), - WalReplayOptions::default(), - table_store, - ) - .await - .unwrap(); - let Err(error) = replay_iter.next().await else { - panic!("missing WAL must fail replay"); - }; - assert!(error.has_object_store_not_found()); - } - - #[tokio::test] - async fn should_remain_failed_after_a_replay_error() { - let table_store = test_table_store(); - for wal_id in [1, 3] { - let mut builder = table_store.wal_table_builder(); - builder - .add(RowEntry::new_value( - format!("key-{wal_id}").as_bytes(), - b"value", - wal_id, - )) - .await - .unwrap(); - let encoded = builder.build().await.unwrap(); - table_store - .write_sst(&SsTableId::Wal(wal_id), &encoded) - .await - .unwrap(); - } - - let mut replay_iter = ExactWalReplayIterator::range( - 1..4, - &ManifestCore::new(), - WalReplayOptions::default(), - table_store, - ) - .await - .unwrap(); - let Err(first) = replay_iter.next().await else { - panic!("missing WAL must fail replay"); - }; - let Err(second) = replay_iter.next().await else { - panic!("failed replay iterator must remain failed"); - }; - assert!(first.has_object_store_not_found()); - assert!(second.has_object_store_not_found()); - } - - #[tokio::test] - async fn should_construct_huge_sparse_range_without_materializing_every_wal_id() { - let inner: Arc = Arc::new(InMemory::new()); - let gated = Arc::new(GatedObjectStore::new(inner)); - gated.get_opts_gate.close(); - let table_store = test_table_store_with_object_store(gated.clone()); - - let replay_iter = tokio::time::timeout( - Duration::from_secs(1), - ExactWalReplayIterator::range( - 1..u64::MAX, - &ManifestCore::new(), - WalReplayOptions { - prefetch: WalReplaySettings { - max_concurrent_objects: 2, - max_inflight_bytes: 2, - }, - ..WalReplayOptions::default() - }, - table_store, - ), - ) - .await - .expect("replay range construction allocated proportional to the WAL ID span") - .unwrap(); - - gated.get_opts_gate.wait_for_arrivals(1).await; - assert_eq!(gated.get_opts_gate.arrivals(), 1); - drop(replay_iter); - } - - #[tokio::test] - async fn should_fail_if_wal_size_changes_after_listing() { - let inner = Arc::new(InMemory::new()); - let gated = Arc::new(GatedObjectStore::new(inner.clone())); - let table_store = test_table_store_with_object_store(gated.clone()); - let mut builder = table_store.wal_table_builder(); - builder - .add(RowEntry::new_value(b"key", b"value", 1)) - .await - .unwrap(); - let encoded = builder.build().await.unwrap(); - table_store - .write_sst(&SsTableId::Wal(1), &encoded) - .await - .unwrap(); - gated.get_opts_gate.close(); - - let mut replay_iter = ExactWalReplayIterator::range( - 1..2, - &ManifestCore::new(), - WalReplayOptions::default(), - table_store, - ) - .await - .unwrap(); - gated.get_opts_gate.wait_for_arrivals(1).await; - let location = Path::from("/tmp/test_kv_store/wal/00000000000000000001.sst"); - inner - .put(&location, Bytes::from_static(b"changed").into()) - .await - .unwrap(); - gated.get_opts_gate.release(); - - let Err(error) = replay_iter.next().await else { - panic!("changed WAL size must fail replay"); - }; - assert!(matches!(error, SlateDBError::WalDataError(_))); - } - - #[tokio::test] - async fn should_prefetch_up_to_the_object_limit() { - let inner: Arc = Arc::new(InMemory::new()); - let gated = Arc::new(GatedObjectStore::new(inner)); - let table_store = test_table_store_with_object_store(gated.clone()); - for wal_id in 1..=4 { - let mut builder = table_store.wal_table_builder(); - builder - .add(RowEntry::new_value( - format!("key-{wal_id}").as_bytes(), - b"value", - wal_id, - )) - .await - .unwrap(); - let encoded = builder.build().await.unwrap(); - table_store - .write_sst(&SsTableId::Wal(wal_id), &encoded) - .await - .unwrap(); - } - gated.get_opts_gate.close(); - - let mut replay_iter = ExactWalReplayIterator::range( - 1..5, - &ManifestCore::new(), - WalReplayOptions { - prefetch: WalReplaySettings { - max_concurrent_objects: 2, - max_inflight_bytes: 1024 * 1024, - }, - ..WalReplayOptions::default() - }, - table_store, - ) - .await - .unwrap(); - gated.get_opts_gate.wait_for_arrivals(2).await; - assert_eq!(gated.get_opts_gate.arrivals(), 2); - - gated.get_opts_gate.release(); - while replay_iter.next().await.unwrap().is_some() {} - assert_eq!(gated.get_opts_gate.arrivals(), 4); - } - - #[cfg(any( - feature = "snappy", - feature = "zlib", - feature = "lz4", - feature = "zstd" - ))] - #[tokio::test] - async fn compressed_wals_cannot_expand_past_total_replay_memory_budget() { - let mut codecs = Vec::new(); - #[cfg(feature = "snappy")] - codecs.push(crate::config::CompressionCodec::Snappy); - #[cfg(feature = "zlib")] - codecs.push(crate::config::CompressionCodec::Zlib); - #[cfg(feature = "lz4")] - codecs.push(crate::config::CompressionCodec::Lz4); - #[cfg(feature = "zstd")] - codecs.push(crate::config::CompressionCodec::Zstd); - - for codec in codecs { - let inner = Arc::new(InMemory::new()); - let recording = Arc::new(RecordingObjectStore::new(inner)); - let format = SsTableFormat { - compression_codec: Some(codec), - ..SsTableFormat::default() - }; - let table_store = - test_table_store_with_format_and_object_store(format, recording.clone()); - let mut builder = table_store.wal_table_builder(); - builder - .add(RowEntry::new_value(b"large", &vec![b'x'; 1024 * 1024], 1)) - .await - .unwrap(); - let encoded = builder.build().await.unwrap(); - table_store - .write_sst(&SsTableId::Wal(1), &encoded) - .await - .unwrap(); - recording.clear(); - - let total_budget = 256 * 1024; - let mut replay = ExactWalReplayIterator::range( - 1..2, - &ManifestCore::new(), - WalReplayOptions { - prefetch: WalReplaySettings { - max_concurrent_objects: 64, - max_inflight_bytes: total_budget, - }, - ..WalReplayOptions::default() - }, - table_store, - ) - .await - .unwrap(); - assert!(replay.peak_reserved_bytes + replay.working_memory_limit <= total_budget); - let Err(first_error) = replay.next().await else { - panic!("{codec:?} WAL expanded past its memory budget"); - }; - let Err(second_error) = replay.next().await else { - panic!("{codec:?} memory-limit failure was not sticky"); - }; - assert!(matches!( - first_error, - SlateDBError::WalReplayMemoryLimitExceeded { .. } - )); - assert!(matches!( - second_error, - SlateDBError::WalReplayMemoryLimitExceeded { .. } - )); - assert_eq!(recording.get_kinds(false).len(), 1, "{codec:?}"); - } - } - - #[tokio::test] - async fn oversized_encoded_wal_uses_bounded_range_replay() { - let inner = Arc::new(InMemory::new()); - let recording = Arc::new(RecordingObjectStore::new(inner)); - let table_store = test_table_store_with_object_store(recording.clone()); - let mut builder = table_store.wal_table_builder(); - for seq in 1..=64 { - builder - .add(RowEntry::new_value( - format!("large-{seq:03}").as_bytes(), - &[b'x'; 2048], - seq, - )) - .await - .unwrap(); - } - let encoded = builder.build().await.unwrap(); - table_store - .write_sst(&SsTableId::Wal(1), &encoded) - .await - .unwrap(); - recording.clear(); - - let mut replay = ExactWalReplayIterator::range( - 1..2, - &ManifestCore::new(), - WalReplayOptions { - prefetch: WalReplaySettings { - max_concurrent_objects: 64, - max_inflight_bytes: 64 * 1024, - }, - ..WalReplayOptions::default() - }, - table_store, - ) - .await - .unwrap(); - let replayed = replay - .next() - .await - .expect("oversized WAL should remain recoverable") - .expect("oversized WAL should produce a replayed table"); - let mut rows = replayed.table.table().iter(); - let mut row_count = 0; - while let Some(row) = rows.next().await.unwrap() { - assert_eq!(row.value.len(), 2048); - row_count += 1; - } - assert_eq!(row_count, 64); - assert!(replay.next().await.unwrap().is_none()); - assert!(recording.get_kinds(false).len() > 1); - assert!( - recording - .get_range_sizes() - .into_iter() - .all(|range_bytes| range_bytes <= 8 * 1024), - "range replay must not fetch a section larger than its working partition" - ); - } - - #[tokio::test] - async fn oversized_range_replay_rejects_large_metadata_before_fetching_it() { - let inner = Arc::new(InMemory::new()); - let recording = Arc::new(RecordingObjectStore::new(inner.clone())); - let table_store = test_table_store_with_object_store(recording.clone()); - let mut builder = table_store.wal_table_builder(); - for seq in 1..=64 { - builder - .add(RowEntry::new_value( - format!("large-metadata-{seq:03}").as_bytes(), - &[b'x'; 2048], - seq, - )) - .await - .unwrap(); - } - let encoded = builder.build().await.unwrap(); - table_store - .write_sst(&SsTableId::Wal(1), &encoded) - .await - .unwrap(); - let metadata = table_store - .list_wal_ssts_for_replay(1..2) - .await - .unwrap() - .pop() - .unwrap() - .metadata; - let mut bytes = inner - .get(&metadata.location) - .await - .unwrap() - .bytes() - .await - .unwrap() - .to_vec(); - let footer_start = bytes.len() - 10; - bytes[footer_start..footer_start + 8].copy_from_slice(&0_u64.to_be_bytes()); - inner - .put(&metadata.location, Bytes::from(bytes).into()) - .await - .unwrap(); - recording.clear(); - - let mut replay = ExactWalReplayIterator::range( - 1..2, - &ManifestCore::new(), - WalReplayOptions { - prefetch: WalReplaySettings { - max_concurrent_objects: 4, - max_inflight_bytes: 64 * 1024, - }, - ..WalReplayOptions::default() - }, - table_store, - ) - .await - .unwrap(); - let Err(error) = replay.next().await else { - panic!("oversized metadata was not rejected"); - }; - - assert!(matches!( - error, - SlateDBError::WalReplayMemoryLimitExceeded { - kind: "encoded and decoded WAL metadata", - .. - } - )); - assert!( - recording - .get_range_sizes() - .into_iter() - .all(|range_bytes| range_bytes <= 8 * 1024), - "metadata validation must reject before an oversized range GET" - ); - } - - #[tokio::test] - async fn oversized_range_replay_rejects_large_index_before_fetching_it() { - let inner = Arc::new(InMemory::new()); - let recording = Arc::new(RecordingObjectStore::new(inner)); - let format = SsTableFormat { - block_size: 128, - ..SsTableFormat::default() - }; - let table_store = test_table_store_with_format_and_object_store(format, recording.clone()); - let mut builder = table_store.wal_table_builder(); - for seq in 1..=1024 { - builder - .add(RowEntry::new_value( - format!("large-index-{seq:04}-{}", "k".repeat(64)).as_bytes(), - &[b'x'; 128], - seq, - )) - .await - .unwrap(); - } - let encoded = builder.build().await.unwrap(); - let index_len = usize::try_from(encoded.info.index_len).unwrap(); - let total_budget = 64 * 1024; - let metadata_memory_limit = total_budget / 8; - assert!(encoded.remaining_len() > total_budget * 3 / 4); - assert!(index_len > metadata_memory_limit); - assert!(encoded.unconsumed_blocks.len() > 100); - table_store - .write_sst(&SsTableId::Wal(1), &encoded) - .await - .unwrap(); - recording.clear(); - - let mut replay = ExactWalReplayIterator::range( - 1..2, - &ManifestCore::new(), - WalReplayOptions { - prefetch: WalReplaySettings { - max_concurrent_objects: 4, - max_inflight_bytes: total_budget, - }, - ..WalReplayOptions::default() - }, - table_store, - ) - .await - .unwrap(); - let Err(error) = replay.next().await else { - panic!("oversized WAL index was not rejected"); - }; - - assert!(matches!( - error, - SlateDBError::WalReplayMemoryLimitExceeded { - kind: "encoded and decoded WAL index", - .. - } - )); - assert!( - recording - .get_range_sizes() - .into_iter() - .all(|range_bytes| range_bytes <= metadata_memory_limit as u64), - "index validation must reject before the oversized index range GET" - ); - } - - #[tokio::test] - async fn oversized_range_replay_supports_codecs_and_block_transformations() { - let mut codecs = Vec::with_capacity(5); - codecs.push(None); - #[cfg(feature = "snappy")] - codecs.push(Some(crate::config::CompressionCodec::Snappy)); - #[cfg(feature = "zlib")] - codecs.push(Some(crate::config::CompressionCodec::Zlib)); - #[cfg(feature = "lz4")] - codecs.push(Some(crate::config::CompressionCodec::Lz4)); - #[cfg(feature = "zstd")] - codecs.push(Some(crate::config::CompressionCodec::Zstd)); - - for codec in codecs { - for transformed in [false, true] { - let inner = Arc::new(InMemory::new()); - let recording = Arc::new(RecordingObjectStore::new(inner)); - let format = SsTableFormat { - compression_codec: codec, - block_transformer: transformed - .then(|| Arc::new(HistoricalXorTransformer) as Arc), - ..SsTableFormat::default() - }; - let table_store = - test_table_store_with_format_and_object_store(format, recording.clone()); - let mut builder = table_store.wal_table_builder(); - let mut test_rng = rng::new_test_rng(None); - for seq in 1..=64 { - let mut value = vec![0_u8; 2048]; - test_rng.fill(value.as_mut_slice()); - builder - .add(RowEntry::new_value( - format!("codec-{seq:03}").as_bytes(), - &value, - seq, - )) - .await - .unwrap(); - } - let encoded = builder.build().await.unwrap(); - assert!(encoded.remaining_len() > 96 * 1024, "{codec:?}"); - table_store - .write_sst(&SsTableId::Wal(1), &encoded) - .await - .unwrap(); - recording.clear(); - - let mut replay = ExactWalReplayIterator::range( - 1..2, - &ManifestCore::new(), - WalReplayOptions { - prefetch: WalReplaySettings { - max_concurrent_objects: 4, - max_inflight_bytes: 128 * 1024, - }, - ..WalReplayOptions::default() - }, - table_store, - ) - .await - .unwrap(); - let replayed = replay.next().await.unwrap().unwrap(); - assert_eq!(replayed.table.metadata().entry_num, 64, "{codec:?}"); - assert!(replay.next().await.unwrap().is_none()); - assert!( - recording.get_kinds(false).len() > 1, - "{codec:?}, transformed={transformed}" - ); - } - } - } - - #[tokio::test] - async fn exhausted_block_is_dropped_before_decoding_the_next_block() { - let format = SsTableFormat { - block_size: 64, - ..SsTableFormat::default() - }; - let table_store = test_table_store_with_format(format); - let mut builder = table_store.wal_table_builder(); - for seq in 1..=3 { - builder - .add(RowEntry::new_value( - format!("key-{seq}").as_bytes(), - &[b'x'; 48], - seq, - )) - .await - .unwrap(); - } - let encoded = builder.build().await.unwrap(); - table_store - .write_sst(&SsTableId::Wal(1), &encoded) - .await - .unwrap(); - let bytes = table_store - .read_wal_sst_bytes(1, None, encoded.remaining_len()) - .await - .unwrap(); - let DecodedWalSst::Data(wal) = table_store - .decode_wal_sst(1, bytes, 1024 * 1024) - .await - .unwrap() - else { - panic!("data WAL decoded as a fence"); - }; - let probe = Arc::new(super::BlockLifetimeProbe::default()); - let mut iterator = - super::WalBlocksIterator::new(Arc::clone(&table_store), *wal, 1024 * 1024, 1024 * 1024) - .unwrap() - .observe_block_lifetimes_with(Arc::clone(&probe)); - - while iterator.next().await.unwrap().is_some() {} - - assert_eq!(probe.active.load(Ordering::SeqCst), 0); - assert_eq!(probe.peak.load(Ordering::SeqCst), 1); - } - - #[tokio::test] - async fn cancelling_wal_block_iteration_releases_the_current_block() { - let format = SsTableFormat { - block_size: 64, - ..SsTableFormat::default() - }; - let table_store = test_table_store_with_format(format); - let mut builder = table_store.wal_table_builder(); - builder - .add(RowEntry::new_value(b"key", &[b'x'; 48], 1)) - .await - .unwrap(); - let encoded = builder.build().await.unwrap(); - table_store - .write_sst(&SsTableId::Wal(1), &encoded) - .await - .unwrap(); - let bytes = table_store - .read_wal_sst_bytes(1, None, encoded.remaining_len()) - .await - .unwrap(); - let DecodedWalSst::Data(wal) = table_store - .decode_wal_sst(1, bytes, 1024 * 1024) - .await - .unwrap() - else { - panic!("data WAL decoded as a fence"); - }; - let probe = Arc::new(super::BlockLifetimeProbe::default()); - let mut iterator = - super::WalBlocksIterator::new(table_store, *wal, 1024 * 1024, 1024 * 1024) - .unwrap() - .observe_block_lifetimes_with(Arc::clone(&probe)); - assert!(iterator.next().await.unwrap().is_some()); - assert_eq!(probe.active.load(Ordering::SeqCst), 1); - - drop(iterator); - - assert_eq!(probe.active.load(Ordering::SeqCst), 0); - assert_eq!(probe.peak.load(Ordering::SeqCst), 1); - } - - #[tokio::test] - async fn decoding_failure_drops_the_exhausted_block_first() { - let format = SsTableFormat { - block_size: 64, - ..SsTableFormat::default() - }; - let table_store = test_table_store_with_format(format); - let mut builder = table_store.wal_table_builder(); - for seq in 1..=3 { - builder - .add(RowEntry::new_value( - format!("key-{seq}").as_bytes(), - &[b'x'; 48], - seq, - )) - .await - .unwrap(); - } - let encoded = builder.build().await.unwrap(); - let original = encoded.remaining_as_bytes(); - let DecodedWalSst::Data(original_wal) = table_store - .decode_wal_sst(1, original.clone(), 1024 * 1024) - .await - .unwrap() - else { - panic!("data WAL decoded as a fence"); - }; - let second_block_offset = original_wal.index.borrow().block_meta().get(1).offset() as usize; - let mut corrupted = original.to_vec(); - corrupted[second_block_offset] ^= 1; - let DecodedWalSst::Data(wal) = table_store - .decode_wal_sst(1, Bytes::from(corrupted), 1024 * 1024) - .await - .unwrap() - else { - panic!("data WAL decoded as a fence"); - }; - let probe = Arc::new(super::BlockLifetimeProbe::default()); - let mut iterator = - super::WalBlocksIterator::new(table_store, *wal, 1024 * 1024, 1024 * 1024) - .unwrap() - .observe_block_lifetimes_with(Arc::clone(&probe)); - - assert!(iterator.next().await.unwrap().is_some()); - let Err(error) = iterator.next().await else { - panic!("corrupt second block did not fail"); - }; - - assert!(matches!(error, SlateDBError::ChecksumMismatch { .. })); - assert_eq!(probe.active.load(Ordering::SeqCst), 0); - assert_eq!(probe.peak.load(Ordering::SeqCst), 1); - } - - #[tokio::test] - async fn oversized_range_replay_never_exposes_a_partial_wal_after_late_corruption() { - let inner = Arc::new(InMemory::new()); - let table_store = test_table_store_with_object_store(inner.clone()); - let mut builder = table_store.wal_table_builder(); - for seq in 1..=64 { - builder - .add(RowEntry::new_value( - format!("corrupt-{seq:03}").as_bytes(), - &[b'x'; 2048], - seq, - )) - .await - .unwrap(); + Self::range( + wal_id_range, + db_state, + SlateDbWalIteratorOptions::default(), + options, + table_store, + wal_store, + ) } - let encoded = builder.build().await.unwrap(); - assert!(encoded.remaining_len() > 96 * 1024); - table_store - .write_sst(&SsTableId::Wal(1), &encoded) - .await - .unwrap(); - let handle = table_store.open_sst(&SsTableId::Wal(1)).await.unwrap(); - let index = table_store.read_index(&handle, false).await.unwrap(); - let second_block_offset = index.borrow().block_meta().get(1).offset() as usize; - let metadata = table_store - .list_wal_ssts_for_replay(1..2) - .await - .unwrap() - .pop() - .unwrap() - .metadata; - let mut bytes = inner - .get(&metadata.location) - .await - .unwrap() - .bytes() - .await - .unwrap() - .to_vec(); - bytes[second_block_offset] ^= 1; - inner - .put(&metadata.location, Bytes::from(bytes).into()) - .await - .unwrap(); + } - let mut replay = ExactWalReplayIterator::range( - 1..2, + #[tokio::test] + async fn should_return_replayed_rows_before_repeating_terminal_error() { + let table_store = test_table_store(); + let first_row = RowEntry::new_value(b"key_001", b"value_001", 1); + let later_row = RowEntry::new_value(b"key_002", b"value_002", 2); + let wal_iter = ScriptedWalIterator { + results: VecDeque::from([ + Ok(Some(WalRows { + rows: vec![first_row], + last_consumed_wal_file_id: 1, + })), + Err(WalError::WalTruncated(2)), + // A terminal error must prevent the replay iterator from resuming + // the underlying iterator on later calls. + Ok(Some(WalRows { + rows: vec![later_row], + last_consumed_wal_file_id: 3, + })), + ]), + }; + let mut replay_iter = WalReplayIterator::for_wal_iterator( + Box::new(wal_iter), &ManifestCore::new(), WalReplayOptions { - prefetch: WalReplaySettings { - max_concurrent_objects: 4, - max_inflight_bytes: 128 * 1024, - }, + max_memtable_bytes: usize::MAX, ..WalReplayOptions::default() }, - table_store, + Arc::clone(&table_store), ) - .await .unwrap(); - let Err(first) = replay.next().await else { - panic!("corrupt oversized WAL was partially exposed"); - }; - let Err(second) = replay.next().await else { - panic!("corrupt oversized WAL failure was not sticky"); - }; - assert!(matches!(first, SlateDBError::ChecksumMismatch { .. })); - assert!(matches!(second, SlateDBError::ChecksumMismatch { .. })); - } - #[tokio::test] - async fn unbounded_transform_is_rejected_before_decode() { - let decode_called = Arc::new(AtomicBool::new(false)); - let format = SsTableFormat { - block_transformer: Some(Arc::new(UnboundedTransformer { - decode_called: Arc::clone(&decode_called), - })), - ..SsTableFormat::default() - }; - let table_store = test_table_store_with_format(format); - let mut builder = table_store.wal_table_builder(); - builder - .add(RowEntry::new_value(b"key", b"value", 1)) - .await - .unwrap(); - let encoded = builder.build().await.unwrap(); - table_store - .write_sst(&SsTableId::Wal(1), &encoded) - .await - .unwrap(); + let replayed = replay_iter.next().await.unwrap().unwrap(); + assert_eq!(replayed.last_wal_id, 1); + assert_eq!(replayed.last_seq, 1); + assert_eq!(replayed.table.metadata().entry_num, 1); - let mut replay = ExactWalReplayIterator::range( - 1..2, - &ManifestCore::new(), - WalReplayOptions::default(), - table_store, - ) - .await - .unwrap(); assert!(matches!( - replay.next().await, - Err(SlateDBError::BlockTransformError) + replay_iter.next().await, + Err(SlateDBError::WalTruncated(2)) + )); + assert!(matches!( + replay_iter.next().await, + Err(SlateDBError::WalTruncated(2)) )); - assert!(!decode_called.load(Ordering::SeqCst)); } #[tokio::test] - async fn later_checksum_valid_structural_corruption_returns_no_partial_wal() { - let inner = Arc::new(InMemory::new()); - let recording = Arc::new(RecordingObjectStore::new(inner.clone())); - let table_store = test_table_store_with_object_store(recording.clone()); - let mut builder = table_store.wal_table_builder(); - for seq in 1..=12 { - builder - .add(RowEntry::new_value( - format!("key-{seq:02}").as_bytes(), - &[b'x'; 1024], - seq, - )) - .await - .unwrap(); - } - let encoded = builder.build().await.unwrap(); - let mut bytes = encoded.remaining_as_bytes().to_vec(); - let decoded = table_store - .decode_wal_sst(1, Bytes::copy_from_slice(&bytes), 64 * 1024 * 1024) - .await - .unwrap(); - let DecodedWalSst::Data(wal) = decoded else { - panic!("data WAL decoded as a fence"); - }; - let block_meta = wal.index.borrow().block_meta(); - assert!(block_meta.len() >= 2); - let block_index = block_meta.len() - 1; - let start = block_meta.get(block_index).offset() as usize; - let end = wal.info.filter_offset as usize; - drop(wal); - let payload_end = end - crate::format::sst::CHECKSUM_SIZE; - bytes[payload_end - 2..payload_end].copy_from_slice(&0u16.to_be_bytes()); - let checksum = crc32fast::hash(&bytes[start..payload_end]); - bytes[payload_end..end].copy_from_slice(&checksum.to_be_bytes()); - let path = Path::from("/tmp/test_kv_store/wal/00000000000000000001.sst"); - inner.put(&path, Bytes::from(bytes).into()).await.unwrap(); - recording.clear(); - - let mut replay = ExactWalReplayIterator::range( - 1..2, + async fn should_repeat_terminal_none_for_wal_replay_iterator() { + let table_store = test_table_store(); + let wal_iter = ScriptedWalIterator { + results: VecDeque::from([ + Ok(None), + // Normal termination must prevent the replay iterator from + // resuming the underlying iterator on later calls. + Ok(Some(WalRows { + rows: vec![RowEntry::new_value(b"key", b"value", 1)], + last_consumed_wal_file_id: 1, + })), + ]), + }; + let mut replay_iter = WalReplayIterator::for_wal_iterator( + Box::new(wal_iter), &ManifestCore::new(), WalReplayOptions::default(), - table_store, + Arc::clone(&table_store), ) - .await .unwrap(); - assert!(replay.next().await.is_err()); - assert!(replay.next().await.is_err()); - assert_eq!(recording.get_kinds(false).len(), 1); - } - - #[tokio::test] - async fn checksum_valid_metadata_index_block_and_row_mutations_never_panic() { - let table_store = test_table_store(); - let mut builder = table_store.wal_table_builder(); - for seq in 1..=16 { - builder - .add(RowEntry::new( - format!("key-{seq:02}").into(), - if seq % 3 == 0 { - ValueDeletable::Merge(Bytes::from_static(b"merge")) - } else if seq % 5 == 0 { - ValueDeletable::Tombstone - } else { - ValueDeletable::Value(Bytes::from(vec![seq as u8; 256])) - }, - seq, - (seq % 2 == 0).then_some(seq as i64), - (seq % 4 == 0).then_some((seq * 10) as i64), - )) - .await - .unwrap(); - } - let encoded = builder.build().await.unwrap().remaining_as_bytes(); - let decoded = table_store - .decode_wal_sst(1, encoded.clone(), 64 * 1024 * 1024) - .await - .unwrap(); - let DecodedWalSst::Data(wal) = decoded else { - panic!("data WAL decoded as a fence"); - }; - let metadata_start = u64::from_be_bytes( - encoded[encoded.len() - 10..encoded.len() - 2] - .try_into() - .unwrap(), - ) as usize; - let metadata_end = encoded.len() - 10; - let index_start = wal.info.index_offset as usize; - let index_end = index_start + wal.info.index_len as usize; - let block_meta = wal.index.borrow().block_meta(); - let block_start = block_meta.get(0).offset() as usize; - let block_end = if block_meta.len() > 1 { - block_meta.get(1).offset() as usize - } else { - wal.info.filter_offset as usize - }; - drop(wal); - - let sections = [ - ("metadata", metadata_start, metadata_end), - ("index", index_start, index_end), - ("block-and-row", block_start, block_end), - ]; - for (section_name, start, end) in sections { - let payload_end = end - crate::format::sst::CHECKSUM_SIZE; - assert!(payload_end > start); - let mut structural_errors = 0; - for mutation in 0..256usize { - let mut mutated = encoded.to_vec(); - let position = start + mutation % (payload_end - start); - mutated[position] ^= 1u8 << (mutation % 8); - let checksum = crc32fast::hash(&mutated[start..payload_end]); - mutated[payload_end..end].copy_from_slice(&checksum.to_be_bytes()); - let table_store = Arc::clone(&table_store); - let decode = std::panic::AssertUnwindSafe(async move { - let decoded = table_store - .decode_wal_sst(1, Bytes::from(mutated), 64 * 1024 * 1024) - .await?; - let DecodedWalSst::Data(wal) = decoded else { - return Err(SlateDBError::CorruptSst { - reason: "mutated data WAL decoded as a fence", - path: None, - }); - }; - let mut iterator = super::WalBlocksIterator::new( - Arc::clone(&table_store), - *wal, - 192 * 1024 * 1024, - 64 * 1024 * 1024, - )?; - while iterator.next().await?.is_some() {} - Ok::<(), SlateDBError>(()) - }) - .catch_unwind() - .await; - let result = decode.unwrap_or_else(|_| { - panic!("checksum-valid {section_name} mutation {mutation} panicked") - }); - if result.is_err() { - structural_errors += 1; - } - } - assert!( - structural_errors > 0, - "{section_name} mutations never reached a structural error" - ); - } - } - proptest! { - #![proptest_config(ProptestConfig::with_cases(512))] - - #[test] - fn checksum_valid_wal_structural_fuzz_never_panics( - section in 0usize..3, - mutations in prop::collection::vec((any::(), any::()), 1..=16), - ) { - let (_, encoded) = historical_wal_fixtures() - .into_iter() - .find(|(name, _)| *name == "v2-none") - .unwrap(); - let table_store = test_table_store(); - let decoded = futures::executor::block_on( - table_store.decode_wal_sst(1, encoded.clone(), 64 * 1024 * 1024), - ) - .unwrap(); - let DecodedWalSst::Data(wal) = decoded else { - panic!("historical data WAL decoded as a fence"); - }; - let metadata_start = u64::from_be_bytes( - encoded[encoded.len() - 10..encoded.len() - 2] - .try_into() - .unwrap(), - ) as usize; - let metadata_end = encoded.len() - 10; - let index_start = wal.info.index_offset as usize; - let index_end = index_start + wal.info.index_len as usize; - let block_meta = wal.index.borrow().block_meta(); - let block_start = block_meta.get(0).offset() as usize; - let block_end = if block_meta.len() > 1 { - block_meta.get(1).offset() as usize - } else { - wal.info.filter_offset as usize - }; - let sections = [ - (metadata_start, metadata_end), - (index_start, index_end), - (block_start, block_end), - ]; - let (start, end) = sections[section]; - let payload_end = end - crate::format::sst::CHECKSUM_SIZE; - let mut mutated = encoded.to_vec(); - for (offset, value) in mutations { - let position = start + offset % (payload_end - start); - mutated[position] ^= value.max(1); - } - let checksum = crc32fast::hash(&mutated[start..payload_end]); - mutated[payload_end..end].copy_from_slice(&checksum.to_be_bytes()); - - let decode = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { - futures::executor::block_on(async { - let decoded = table_store - .decode_wal_sst(1, Bytes::from(mutated), 64 * 1024 * 1024) - .await?; - let DecodedWalSst::Data(wal) = decoded else { - return Err(SlateDBError::CorruptSst { - reason: "mutated data WAL decoded as a fence", - path: None, - }); - }; - let mut iterator = super::WalBlocksIterator::new( - Arc::clone(&table_store), - *wal, - 192 * 1024 * 1024, - 64 * 1024 * 1024, - )?; - while iterator.next().await?.is_some() {} - Ok::<(), SlateDBError>(()) - }) - })); - prop_assert!(decode.is_ok()); - } + assert!(replay_iter.next().await.unwrap().is_none()); + assert!(replay_iter.next().await.unwrap().is_none()); } #[tokio::test] - async fn should_apply_wals_in_id_order_when_later_get_returns_first() { - let inner: Arc = Arc::new(InMemory::new()); - let first_path = Path::from("/tmp/test_kv_store/wal/00000000000000000001.sst"); - let reordered = Arc::new(PausedFirstWalGetStore::new(inner, first_path)); - let table_store = test_table_store_with_object_store(reordered.clone()); - for wal_id in 1..=2 { - let mut builder = table_store.wal_table_builder(); - builder - .add(RowEntry::new_value( - format!("key-{wal_id}").as_bytes(), - &[b'x'; 128], - wal_id, - )) - .await - .unwrap(); - let encoded = builder.build().await.unwrap(); - table_store - .write_sst(&SsTableId::Wal(wal_id), &encoded) - .await - .unwrap(); - } - - let mut replay_iter = ExactWalReplayIterator::range( - 1..3, + async fn should_use_last_consumed_wal_file_id_as_replay_watermark() { + let table_store = test_table_store(); + let first_row = RowEntry::new_value(b"key_001", &[b'x'; 128], 1); + let second_row = RowEntry::new_value(b"key_002", &[b'x'; 128], 2); + let max_memtable_bytes = + table_store.estimate_encoded_size_compacted(1, first_row.estimated_size()); + let wal_iter = ScriptedWalIterator { + results: VecDeque::from([ + Ok(Some(WalRows { + rows: vec![first_row], + // The first batch ends partway through WAL file 1. + last_consumed_wal_file_id: 0, + })), + Ok(Some(WalRows { + rows: vec![second_row], + // The second batch consumes the rest of WAL file 1. + last_consumed_wal_file_id: 1, + })), + ]), + }; + let mut replay_iter = WalReplayIterator::for_wal_iterator( + Box::new(wal_iter), &ManifestCore::new(), WalReplayOptions { - prefetch: WalReplaySettings { - max_concurrent_objects: 2, - max_inflight_bytes: 1024 * 1024, - }, - max_memtable_bytes: 1, + max_memtable_bytes, ..WalReplayOptions::default() }, - table_store, + Arc::clone(&table_store), ) - .await .unwrap(); - PausedFirstWalGetStore::wait_until(&reordered.first_started).await; - PausedFirstWalGetStore::wait_until(&reordered.later_get_returned).await; - let (result_tx, mut result_rx) = tokio::sync::mpsc::unbounded_channel(); - let replay_task = tokio::spawn(async move { - while let Some(replayed) = replay_iter.next().await.unwrap() { - result_tx.send(replayed.last_wal_id).unwrap(); - } - }); - for _ in 0..100 { - tokio::task::yield_now().await; - } - assert!(matches!( - result_rx.try_recv(), - Err(tokio::sync::mpsc::error::TryRecvError::Empty) - )); + let first = replay_iter.next().await.unwrap().unwrap(); + assert_eq!(first.last_wal_id, 0); + assert_eq!(first.last_seq, 1); - reordered.release_first.notify_one(); - assert_eq!(result_rx.recv().await, Some(1)); - assert_eq!(result_rx.recv().await, Some(2)); - replay_task.await.unwrap(); + let second = replay_iter.next().await.unwrap().unwrap(); + assert_eq!(second.last_wal_id, 1); + assert_eq!(second.last_seq, 2); + assert!(replay_iter.next().await.unwrap().is_none()); } #[tokio::test] - async fn should_abort_pending_fetches_when_replay_iterator_is_dropped() { - let inner: Arc = Arc::new(InMemory::new()); - let first_path = Path::from("/tmp/test_kv_store/wal/00000000000000000001.sst"); - let paused = Arc::new(PausedFirstWalGetStore::new(inner, first_path)); - let table_store = test_table_store_with_object_store(paused.clone()); - let mut builder = table_store.wal_table_builder(); - builder - .add(RowEntry::new_value(b"key", b"value", 1)) - .await - .unwrap(); - let encoded = builder.build().await.unwrap(); - table_store - .write_sst(&SsTableId::Wal(1), &encoded) - .await - .unwrap(); - - let replay_iter = ExactWalReplayIterator::range( - 1..2, + async fn should_replay_empty_wal() { + let (table_store, wal_store) = test_stores(); + write_empty_wal(1, Arc::clone(&wal_store)).await.unwrap(); + let mut replay_iter = WalReplayIterator::all_wal_ids( &ManifestCore::new(), WalReplayOptions::default(), - table_store, + Arc::clone(&table_store), + Arc::clone(&wal_store), ) .await .unwrap(); - PausedFirstWalGetStore::wait_until(&paused.first_started).await; - drop(replay_iter); - tokio::time::timeout( - Duration::from_secs(1), - PausedFirstWalGetStore::wait_until(&paused.first_cancelled), - ) - .await - .expect("pending WAL fetch was detached instead of aborted"); + let Some(table) = replay_iter.next().await.unwrap() else { + panic!("Expected empty table to be returned from iterator") + }; + + assert_eq!(table.last_wal_id, 1); + assert_eq!(table.last_seq, 0); + assert!(table.table.is_empty()); + assert_eq!(table.last_tick, i64::MIN); + assert!(replay_iter.next().await.unwrap().is_none()); } #[tokio::test] - async fn should_abort_later_fetches_after_a_replay_error() { - let inner = Arc::new(InMemory::new()); - let paused_path = Path::from("/tmp/test_kv_store/wal/00000000000000000002.sst"); - let paused = Arc::new(PausedFirstWalGetStore::new(inner.clone(), paused_path)); - let table_store = test_table_store_with_object_store(paused.clone()); - let mut first_wal_location = None; - for wal_id in 1..=2 { - let mut builder = table_store.wal_table_builder(); - builder - .add(RowEntry::new_value( - format!("key-{wal_id}").as_bytes(), - b"value", - wal_id, - )) - .await - .unwrap(); - let encoded = builder.build().await.unwrap(); - table_store - .write_sst(&SsTableId::Wal(wal_id), &encoded) - .await - .unwrap(); - if wal_id == 1 { - first_wal_location = Some( - table_store - .list_wal_ssts_for_replay(1..2) - .await - .unwrap() - .pop() - .unwrap() - .metadata - .location, - ); - } - } - let first_wal_location = first_wal_location.unwrap(); - let mut corrupted = inner - .get(&first_wal_location) - .await - .unwrap() - .bytes() - .await - .unwrap() - .to_vec(); - corrupted[0] ^= 1; - inner - .put(&first_wal_location, corrupted.into()) - .await - .unwrap(); - - let mut replay_iter = ExactWalReplayIterator::range( - 1..3, + async fn should_replay_zero_byte_wal_fence() { + let (table_store, wal_store) = test_stores(); + wal_store.write_wal_fence(1).await.unwrap(); + let mut replay_iter = WalReplayIterator::all_wal_ids( &ManifestCore::new(), - WalReplayOptions { - prefetch: WalReplaySettings { - max_concurrent_objects: 2, - max_inflight_bytes: 1024 * 1024, - }, - ..WalReplayOptions::default() - }, - table_store, + WalReplayOptions::default(), + Arc::clone(&table_store), + Arc::clone(&wal_store), ) .await .unwrap(); - PausedFirstWalGetStore::wait_until(&paused.first_started).await; - assert!(matches!( - replay_iter.next().await, - Err(SlateDBError::ChecksumMismatch { .. }) - )); - tokio::time::timeout( - Duration::from_secs(1), - PausedFirstWalGetStore::wait_until(&paused.first_cancelled), - ) - .await - .expect("a replay error did not abort a later in-flight WAL fetch"); + let Some(table) = replay_iter.next().await.unwrap() else { + panic!("Expected empty table to be returned from iterator") + }; + + assert_eq!(table.last_wal_id, 1); + assert_eq!(table.last_seq, 0); + assert!(table.table.is_empty()); + assert_eq!(table.last_tick, i64::MIN); + assert!(replay_iter.next().await.unwrap().is_none()); } #[tokio::test] - async fn should_fail_terminally_and_abort_later_fetches_after_a_fetch_panic() { - let inner: Arc = Arc::new(InMemory::new()); - let first_path = Path::from("/tmp/test_kv_store/wal/00000000000000000001.sst"); - let paused_path = Path::from("/tmp/test_kv_store/wal/00000000000000000002.sst"); - let store = Arc::new(PausedFirstWalGetStore::new(inner, paused_path).panic_on(first_path)); - let table_store = test_table_store_with_object_store(store.clone()); - for wal_id in 1..=2 { - let mut builder = table_store.wal_table_builder(); - builder - .add(RowEntry::new_value( - format!("key-{wal_id}").as_bytes(), - b"value", - wal_id, - )) - .await - .unwrap(); - let encoded = builder.build().await.unwrap(); - table_store - .write_sst(&SsTableId::Wal(wal_id), &encoded) - .await - .unwrap(); - } + async fn should_replay_zero_byte_wal_fence_before_real_wal() { + let (table_store, wal_store) = test_stores(); + wal_store.write_wal_fence(1).await.unwrap(); + + let row = RowEntry::new_value(b"key", b"value", 1); + let mut builder = wal_store.table_builder(); + builder.add(row.clone()).await.unwrap(); + let encoded_sst = builder.build().await.unwrap(); + wal_store.write_sst(2, &encoded_sst).await.unwrap(); - let mut replay_iter = ExactWalReplayIterator::range( - 1..3, + let mut replay_iter = WalReplayIterator::all_wal_ids( &ManifestCore::new(), - WalReplayOptions { - prefetch: WalReplaySettings { - max_concurrent_objects: 2, - max_inflight_bytes: 1024 * 1024, - }, - ..WalReplayOptions::default() - }, - table_store, + WalReplayOptions::default(), + Arc::clone(&table_store), + Arc::clone(&wal_store), ) .await .unwrap(); - PausedFirstWalGetStore::wait_until(&store.first_started).await; - let Err(first) = replay_iter.next().await else { - panic!("panicking WAL fetch must fail replay"); - }; - let Err(second) = replay_iter.next().await else { - panic!("failed replay iterator must remain failed"); + let Some(replayed_table) = replay_iter.next().await.unwrap() else { + panic!("Expected table to be returned from iterator") }; - assert!(matches!(first, SlateDBError::BackgroundTaskPanic(_))); - assert!(matches!(second, SlateDBError::BackgroundTaskPanic(_))); - tokio::time::timeout( - Duration::from_secs(1), - PausedFirstWalGetStore::wait_until(&store.first_cancelled), - ) - .await - .expect("fetch panic did not abort a later in-flight WAL fetch"); + assert_eq!(replayed_table.last_wal_id, 2); + assert_eq!(replayed_table.last_seq, 1); + + let mut iter = replayed_table.table.table().iter(); + test_utils::assert_iterator(&mut iter, vec![row]).await; + assert!(replay_iter.next().await.unwrap().is_none()); } #[tokio::test] - async fn should_fail_terminally_if_a_fetch_task_is_cancelled() { - let inner: Arc = Arc::new(InMemory::new()); - let gated = Arc::new(GatedObjectStore::new(inner)); - let table_store = test_table_store_with_object_store(gated.clone()); - let mut builder = table_store.wal_table_builder(); - builder - .add(RowEntry::new_value(b"key", b"value", 1)) - .await - .unwrap(); - let encoded = builder.build().await.unwrap(); - table_store - .write_sst(&SsTableId::Wal(1), &encoded) + async fn should_replay_all_entries() { + let (table_store, wal_store) = test_stores(); + let mut rng = rng::new_test_rng(None); + let entries = sample::table(&mut rng, 1000, 10); + let next_wal_id = write_wals(&entries, 1, &mut rng, 200, Arc::clone(&wal_store)) .await .unwrap(); - gated.get_opts_gate.close(); - let mut replay_iter = ExactWalReplayIterator::range( - 1..2, + let mut replay_iter = WalReplayIterator::all_wal_ids( &ManifestCore::new(), WalReplayOptions::default(), - table_store, + Arc::clone(&table_store), + Arc::clone(&wal_store), ) .await .unwrap(); - gated.get_opts_gate.wait_for_arrivals(1).await; - replay_iter.pending_fetches.front().unwrap().handle.abort(); - let Err(first) = replay_iter.next().await else { - panic!("cancelled WAL fetch must fail replay"); - }; - let Err(second) = replay_iter.next().await else { - panic!("failed replay iterator must remain failed"); + let Some(replayed_table) = replay_iter.next().await.unwrap() else { + panic!("Expected table to be returned from iterator") }; - assert!(matches!(first, SlateDBError::BackgroundTaskCancelled(_))); - assert!(matches!(second, SlateDBError::BackgroundTaskCancelled(_))); - } + assert_eq!(replayed_table.last_wal_id + 1, next_wal_id); - #[tokio::test] - async fn should_hold_byte_permits_until_the_wal_is_consumed() { - let inner: Arc = Arc::new(InMemory::new()); - let gated = Arc::new(GatedObjectStore::new(inner)); - let table_store = test_table_store_with_format_and_object_store( - SsTableFormat { - block_size: 256, - ..SsTableFormat::default() - }, - gated.clone(), - ); - for wal_id in 1..=2 { - let mut builder = table_store.wal_table_builder(); - for row_id in 0..16 { - builder - .add(RowEntry::new_value( - format!("key-{wal_id}-{row_id:02}").as_bytes(), - &[b'x'; 128], - wal_id * 16 + row_id, - )) - .await - .unwrap(); - } - let encoded = builder.build().await.unwrap(); - table_store - .write_sst(&SsTableId::Wal(wal_id), &encoded) - .await - .unwrap(); - } - let listed = table_store.list_wal_ssts_for_replay(1..3).await.unwrap(); - let largest_object = listed - .iter() - .map(|entry| usize::try_from(entry.metadata.size).unwrap()) - .max() - .unwrap(); - let total_budget = largest_object.checked_mul(2).unwrap(); - let working_budget = total_budget / 4; - let encoded_budget = total_budget - working_budget; - assert!(encoded_budget >= largest_object); - assert!(encoded_budget < largest_object * 2); - gated.get_opts_gate.close(); - - let mut replay_iter = ExactWalReplayIterator::range( - 1..3, - &ManifestCore::new(), - WalReplayOptions { - prefetch: WalReplaySettings { - max_concurrent_objects: 2, - max_inflight_bytes: total_budget, - }, - max_memtable_bytes: 1, - ..WalReplayOptions::default() - }, - table_store, + let mut imm_table_iter = replayed_table.table.table().iter(); + test_utils::assert_ranged_kv_scan( + &entries, + &BytesRange::from(..), + IterationOrder::Ascending, + &mut imm_table_iter, ) - .await - .unwrap(); - gated.get_opts_gate.wait_for_arrivals(1).await; - for _ in 0..100 { - tokio::task::yield_now().await; - } - assert_eq!(gated.get_opts_gate.arrivals(), 1); - - gated.get_opts_gate.release(); - assert!(replay_iter.next().await.unwrap().is_some()); - assert!(replay_iter.next().await.unwrap().is_some()); + .await; assert!(replay_iter.next().await.unwrap().is_none()); - assert_eq!(gated.get_opts_gate.arrivals(), 2); } #[tokio::test] async fn should_enforce_max_memtable_bytes() { - let table_store = test_table_store(); + let (table_store, wal_store) = test_stores(); let mut rng = rng::new_test_rng(None); let num_entries = 5000; let entries = sample::table(&mut rng, num_entries, 10); - let next_wal_id = write_wals(&entries, 1, &mut rng, 200, Arc::clone(&table_store)) + let next_wal_id = write_wals(&entries, 1, &mut rng, 200, Arc::clone(&wal_store)) .await .unwrap(); let max_memtable_bytes = 1024; - let mut replay_iter = ExactWalReplayIterator::all_wal_ids( + let mut replay_iter = WalReplayIterator::all_wal_ids( &ManifestCore::new(), WalReplayOptions { max_memtable_bytes, ..WalReplayOptions::default() }, Arc::clone(&table_store), + Arc::clone(&wal_store), ) .await .unwrap(); @@ -3603,7 +536,7 @@ mod tests { #[tokio::test] async fn should_apply_max_memtable_bytes_at_wal_boundaries() { - let table_store = test_table_store(); + let (table_store, wal_store) = test_stores(); let wal_entries = [ vec![RowEntry::new_value(b"key_001", &[b'x'; 128], 1)], vec![RowEntry::new_value(b"key_002", &[b'x'; 128], 2)], @@ -3614,24 +547,25 @@ mod tests { table_store.estimate_encoded_size_compacted(1, single_row_size) + 1; for (wal_id, entries) in wal_entries.into_iter().enumerate() { - let mut builder = table_store.wal_table_builder(); + let mut builder = wal_store.table_builder(); for entry in entries { builder.add(entry).await.unwrap(); } let encoded_sst = builder.build().await.unwrap(); - table_store - .write_sst(&SsTableId::Wal(wal_id as u64 + 1), &encoded_sst) + wal_store + .write_sst(wal_id as u64 + 1, &encoded_sst) .await .unwrap(); } - let mut replay_iter = ExactWalReplayIterator::all_wal_ids( + let mut replay_iter = WalReplayIterator::all_wal_ids( &ManifestCore::new(), WalReplayOptions { max_memtable_bytes, ..WalReplayOptions::default() }, Arc::clone(&table_store), + Arc::clone(&wal_store), ) .await .unwrap(); @@ -3663,7 +597,7 @@ mod tests { #[tokio::test] async fn should_not_split_one_commit_seq_across_replayed_memtables() { - let table_store = test_table_store(); + let (table_store, wal_store) = test_stores(); let commit_seq = 42; // Simulate one committed write batch. Every row gets the same commit @@ -3681,25 +615,23 @@ mod tests { table_store.estimate_encoded_size_compacted(1, entries[0].estimated_size()); // Use the real WAL SST builder so the fixture matches WAL flushes. - let mut builder = table_store.wal_table_builder(); + let mut builder = wal_store.table_builder(); for entry in entries { builder.add(entry).await.unwrap(); } let encoded_sst = builder.build().await.unwrap(); - table_store - .write_sst(&SsTableId::Wal(1), &encoded_sst) - .await - .unwrap(); + wal_store.write_sst(1, &encoded_sst).await.unwrap(); // Replay the single WAL SST into in-memory tables. If the replay code // can split a single commit sequence, it will do so here. - let mut replay_iter = ExactWalReplayIterator::all_wal_ids( + let mut replay_iter = WalReplayIterator::all_wal_ids( &ManifestCore::new(), WalReplayOptions { max_memtable_bytes, ..WalReplayOptions::default() }, Arc::clone(&table_store), + Arc::clone(&wal_store), ) .await .unwrap(); @@ -3722,7 +654,7 @@ mod tests { #[tokio::test] async fn should_replay_memtables_in_sequence_order() { - let table_store = test_table_store(); + let (table_store, wal_store) = test_stores(); // Write one WAL with entries whose sequence numbers do not match key // order. Replay must not expose a later memtable whose sequence range @@ -3740,24 +672,22 @@ mod tests { // Use the real WAL SST builder so replay sees the same entry order as a // flushed WAL. - let mut builder = table_store.wal_table_builder(); + let mut builder = wal_store.table_builder(); for entry in entries { builder.add(entry).await.unwrap(); } let encoded_sst = builder.build().await.unwrap(); - table_store - .write_sst(&SsTableId::Wal(1), &encoded_sst) - .await - .unwrap(); + wal_store.write_sst(1, &encoded_sst).await.unwrap(); // Replay the single WAL SST into in-memory tables. - let mut replay_iter = ExactWalReplayIterator::all_wal_ids( + let mut replay_iter = WalReplayIterator::all_wal_ids( &ManifestCore::new(), WalReplayOptions { max_memtable_bytes, ..WalReplayOptions::default() }, Arc::clone(&table_store), + Arc::clone(&wal_store), ) .await .unwrap(); @@ -3782,7 +712,7 @@ mod tests { #[tokio::test] async fn should_only_replay_wals_after_last_l0_flushed_wal_id() { - let table_store = test_table_store(); + let (table_store, wal_store) = test_stores(); let mut rng = rng::new_test_rng(None); let compacted_entries = sample::table(&mut rng, 1000, 10); let mut next_wal_id = 1; @@ -3792,7 +722,7 @@ mod tests { next_wal_id, &mut rng, 200, - Arc::clone(&table_store), + Arc::clone(&wal_store), ) .await .unwrap(); @@ -3804,7 +734,7 @@ mod tests { next_wal_id, &mut rng, 200, - Arc::clone(&table_store), + Arc::clone(&wal_store), ) .await .unwrap(); @@ -3813,10 +743,11 @@ mod tests { db_state.replay_after_wal_id = replay_after_wal_id; db_state.next_wal_sst_id = replay_after_wal_id + 1; - let mut replay_iter = ExactWalReplayIterator::all_wal_ids( + let mut replay_iter = WalReplayIterator::all_wal_ids( &db_state, WalReplayOptions::default(), Arc::clone(&table_store), + Arc::clone(&wal_store), ) .await .unwrap(); @@ -3839,10 +770,10 @@ mod tests { #[tokio::test] async fn should_replay_wals_after_min_seq() { - let table_store = test_table_store(); + let (table_store, wal_store) = test_stores(); let mut rng = rng::new_test_rng(None); let entries = sample::table(&mut rng, 1000, 10); - let next_wal_id = write_wals(&entries, 1, &mut rng, 200, Arc::clone(&table_store)) + let next_wal_id = write_wals(&entries, 1, &mut rng, 200, Arc::clone(&wal_store)) .await .unwrap(); @@ -3852,10 +783,11 @@ mod tests { db_state.last_l0_seq = min_seq; db_state.last_l0_clock_tick = 0; - let mut replay_iter = ExactWalReplayIterator::all_wal_ids( + let mut replay_iter = WalReplayIterator::all_wal_ids( &db_state, WalReplayOptions::default(), Arc::clone(&table_store), + Arc::clone(&wal_store), ) .await .unwrap(); @@ -3877,93 +809,28 @@ mod tests { assert_eq!(total, 500); } - fn historical_wal_fixtures() -> Vec<(&'static str, Bytes)> { - let source = include_str!("../testdata/historical-wal-fixtures.hex"); - source - .lines() - .filter(|line| !line.is_empty() && !line.starts_with('#')) - .filter_map(|line| { - let (name, encoded) = line.split_once(' ')?; - // Each recognized suffix has a different compile-time value; - // this cannot be reduced to one `matches!` expression. - #[allow(clippy::match_like_matches_macro)] - let feature_enabled = match name.rsplit_once('-').map(|(_, suffix)| suffix) { - Some("snappy") => cfg!(feature = "snappy"), - Some("zlib") => cfg!(feature = "zlib"), - Some("lz4") => cfg!(feature = "lz4"), - Some("zstd") => cfg!(feature = "zstd"), - _ => true, - }; - feature_enabled.then(|| (name, Bytes::from(decode_hex(encoded)))) - }) - .collect() - } - - fn decode_hex(encoded: &str) -> Vec { - assert!(encoded.len().is_multiple_of(2)); - encoded - .as_bytes() - .chunks_exact(2) - .map(|pair| { - let high = (pair[0] as char).to_digit(16).unwrap(); - let low = (pair[1] as char).to_digit(16).unwrap(); - ((high << 4) | low) as u8 - }) - .collect() - } - - fn assert_historical_rows(name: &str, rows: &[RowEntry]) { - assert_eq!(rows.len(), 5, "fixture {name}"); - assert_eq!( - rows.iter().map(|row| row.seq).collect::>(), - vec![7, 8, 9, 10, 11] - ); - assert_eq!(rows[0].key, Bytes::from_static(b"alpha")); - assert_eq!( - rows[0].value, - ValueDeletable::Value(Bytes::from_static(b"value-a")) - ); - assert_eq!( - (rows[0].create_ts, rows[0].expire_ts), - (Some(100), Some(200)) - ); - assert_eq!(rows[1].key, Bytes::from_static(b"dup")); - assert_eq!( - rows[1].value, - ValueDeletable::Merge(Bytes::from_static(b"merge-1")) - ); - assert_eq!(rows[2].key, Bytes::from_static(b"dup")); - assert_eq!(rows[3].value, ValueDeletable::Tombstone); - assert_eq!(rows[4].value.len(), 20); - } - fn test_table_store() -> Arc { - let object_store: Arc = Arc::new(InMemory::new()); - test_table_store_with_object_store(object_store) + test_stores().0 } - fn test_table_store_with_object_store(object_store: Arc) -> Arc { - test_table_store_with_format_and_object_store(SsTableFormat::default(), object_store) - } - - fn test_table_store_with_format(format: SsTableFormat) -> Arc { + fn test_stores() -> (Arc, Arc) { let object_store: Arc = Arc::new(InMemory::new()); - test_table_store_with_format_and_object_store(format, object_store) - } - - fn test_table_store_with_format_and_object_store( - format: SsTableFormat, - object_store: Arc, - ) -> Arc { let path = Path::from("/tmp/test_kv_store"); - Arc::new(TableStore::new( - ObjectStores::new(object_store, None), - format, - path, + let table_store = Arc::new(TableStore::new( + ObjectStores::new(Arc::clone(&object_store), None), + SsTableFormat::default(), + path.clone(), None, TableStoreKind::Main, BlockCachePolicy::default(), - )) + )); + let wal_store = Arc::new(WalTableStore::new( + object_store, + SsTableFormat::default(), + path, + TableStoreKind::Main, + )); + (table_store, wal_store) } /// Write a sequence of WALs with a random (bounded) number of entries. @@ -3973,7 +840,7 @@ mod tests { next_wal_id: u64, rng: &mut TestRng, max_wal_entries: usize, - table_store: Arc, + wal_store: Arc, ) -> Result { let mut iter = entries.iter(); let mut next_seq = 1; @@ -3990,7 +857,7 @@ mod tests { next_seq, &mut iter, wal_entries, - Arc::clone(&table_store), + Arc::clone(&wal_store), ) .await?; next_wal_id += 1; @@ -4001,11 +868,11 @@ mod tests { async fn write_empty_wal( wal_id: u64, - table_store: Arc, + wal_store: Arc, ) -> Result<(), SlateDBError> { let empty_entries = BTreeMap::new(); let mut empty_iter = empty_entries.iter(); - let _ = write_wal(wal_id, 0, &mut empty_iter, 0, table_store).await?; + let _ = write_wal(wal_id, 0, &mut empty_iter, 0, wal_store).await?; Ok(()) } @@ -4014,20 +881,22 @@ mod tests { next_seq: u64, entries: &mut Iter<'_, Bytes, Bytes>, max_entries: usize, - table_store: Arc, + wal_store: Arc, ) -> Result { - let mut writer = table_store.table_writer(SsTableId::Wal(wal_id)); + let mut builder = wal_store.table_builder(); let mut next_seq = next_seq; - while next_seq < next_seq + (max_entries as u64) { + let end_seq = next_seq + (max_entries as u64); + while next_seq < end_seq { let Some((key, value)) = entries.next() else { break; }; - writer + builder .add(RowEntry::new_value(key, value, next_seq)) .await?; next_seq += 1; } - writer.close().await?; + let encoded_sst = builder.build().await?; + wal_store.write_sst(wal_id, &encoded_sst).await?; Ok(next_seq) } } diff --git a/slatedb/tests/custom_wal.rs b/slatedb/tests/custom_wal.rs new file mode 100644 index 0000000000..14239a9773 --- /dev/null +++ b/slatedb/tests/custom_wal.rs @@ -0,0 +1,524 @@ +use std::collections::{BTreeMap, VecDeque}; +use std::ops::Bound; +use std::sync::Arc; +use std::time::Duration; + +use async_trait::async_trait; +use parking_lot::Mutex; +use slatedb::admin::{Admin, CloneSourceSpec}; +use slatedb::config::{ + CloseOptions, DbReaderOptions, FlushOptions, FlushType, GarbageCollectorDirectoryOptions, + GarbageCollectorOptions, Settings, +}; +use slatedb::object_store::memory::InMemory; +use slatedb::object_store::path::Path; +use slatedb::object_store::ObjectStore; +use slatedb::wal::{ + FlushResultFuture, WalAdmin, WalError, WalEvent, WalFileRange, WalGc, WalIterator, WalObserver, + WalReader, WalRows, WalStatus, WalStatusListener, WalWriter, WriterInit, WriterInitResult, + WriterManifest, +}; +use slatedb::{Db, DbReader, DbReaderMode, GarbageCollectorBuilder, RowEntry, VersionedManifest}; + +/// A deliberately small WAL implementation used to exercise the public pluggable-WAL API. +/// Every call to `WalWriter::append` inserts one write batch under a new WAL file ID. +#[derive(Clone, Default)] +struct BTreeMapWal { + files: Arc>>>, +} + +impl BTreeMapWal { + fn file_ids(&self) -> Vec { + self.files.lock().keys().copied().collect() + } + + fn snapshot(&self) -> BTreeMap> { + self.files.lock().clone() + } + + fn open_status(&self) -> WalStatus { + let files = self.files.lock(); + let last_flushed_wal_id = files.last_key_value().map(|(id, _)| *id).unwrap_or(0); + let last_flushed_seq = files + .last_key_value() + .and_then(|(_, batch)| batch.iter().map(|row| row.seq).max()); + WalStatus { + closed_reason: None, + estimated_bytes: 0, + last_flushed_wal_id, + last_flushed_seq, + buffered_wal_entries_count: 0, + } + } +} + +struct BTreeMapWalIterator { + batches: VecDeque<(u64, Vec)>, +} + +#[async_trait] +impl WalIterator for BTreeMapWalIterator { + async fn next(&mut self) -> Result, WalError> { + Ok(self.batches.pop_front().map(|(wal_file_id, rows)| WalRows { + rows, + last_consumed_wal_file_id: wal_file_id, + })) + } +} + +#[async_trait] +impl WalReader for BTreeMapWal { + async fn iterator( + &self, + wal_file_id_range: WalFileRange, + ) -> Result, WalError> { + let WalFileRange(start, end) = wal_file_id_range; + let batches = self + .files + .lock() + .range((start, end)) + .map(|(id, batch)| (*id, batch.clone())) + .collect(); + Ok(Box::new(BTreeMapWalIterator { batches })) + } + + async fn last_wal_file_id(&self, replay_after_wal_id: u64) -> Result { + Ok(self + .files + .lock() + .range((Bound::Excluded(replay_after_wal_id), Bound::Unbounded)) + .next_back() + .map(|(id, _)| *id) + .unwrap_or(replay_after_wal_id)) + } +} + +struct ObserverState { + status: WalStatus, + listeners: Vec, +} + +#[derive(Clone)] +struct BTreeMapWalObserver { + state: Arc>, +} + +impl BTreeMapWalObserver { + fn status(&self) -> Result { + let status = self.state.lock().status.clone(); + if status.closed_reason.is_some() { + Err(status) + } else { + Ok(status) + } + } +} + +impl WalObserver for BTreeMapWalObserver { + fn status(&self) -> Result { + self.status() + } + + fn subscribe(&self, listener: WalStatusListener) -> Result<(), WalError> { + self.state.lock().listeners.push(listener); + Ok(()) + } +} + +struct BTreeMapWalWriter { + wal: BTreeMapWal, + observer: BTreeMapWalObserver, +} + +impl BTreeMapWalWriter { + fn new(wal: BTreeMapWal) -> Self { + let observer = BTreeMapWalObserver { + state: Arc::new(Mutex::new(ObserverState { + status: wal.open_status(), + listeners: Vec::new(), + })), + }; + Self { wal, observer } + } + + fn publish(&self, event: WalEvent) { + let listeners = self.observer.state.lock().listeners.clone(); + for listener in listeners { + listener(event.clone()); + } + } +} + +#[async_trait] +impl WalWriter for BTreeMapWalWriter { + async fn append(&mut self, write_batch: &[RowEntry]) -> Result<(), WalError> { + if self.observer.state.lock().status.closed_reason.is_some() { + return Err(WalError::Closed); + } + + let wal_file_id = { + let mut files = self.wal.files.lock(); + let wal_file_id = match files.last_key_value() { + Some((last_id, _)) => last_id.checked_add(1).ok_or_else(|| { + WalError::InternalError(Arc::new(std::io::Error::other("WAL file ID overflow"))) + })?, + None => 1, + }; + files.insert(wal_file_id, write_batch.to_vec()); + wal_file_id + }; + + let status = { + let mut state = self.observer.state.lock(); + state.status.last_flushed_wal_id = wal_file_id; + state.status.last_flushed_seq = write_batch.iter().map(|row| row.seq).max(); + state.status.clone() + }; + self.publish(WalEvent::WalFlushed(status)); + Ok(()) + } + + async fn flush(&mut self) -> Result { + Ok(Box::pin(async { Ok(()) })) + } + + fn observer(&self) -> Box { + Box::new(self.observer.clone()) + } + + fn status(&self) -> Result { + self.observer.status() + } + + async fn close(&mut self) -> Result<(), WalError> { + let status = { + let mut state = self.observer.state.lock(); + if state.status.closed_reason.is_some() { + return Ok(()); + } + state.status.closed_reason = Some(WalError::Closed); + state.status.clone() + }; + self.publish(WalEvent::WalClosed(status)); + Ok(()) + } +} + +#[async_trait] +impl WriterInit for BTreeMapWal { + async fn fence_and_init( + &self, + manifest: &mut WriterManifest, + ) -> Result { + let replay_after_wal_id = manifest.replay_after_wal_id(); + let start = replay_after_wal_id.checked_add(1).ok_or_else(|| { + WalError::InternalError(Arc::new(std::io::Error::other("WAL replay range overflow"))) + })?; + let end = self + .last_wal_file_id(replay_after_wal_id) + .await? + .checked_add(1) + .ok_or_else(|| { + WalError::InternalError(Arc::new(std::io::Error::other( + "WAL replay range overflow", + ))) + })?; + let replay_iterator = self.iterator((start..end.max(start)).into()).await?; + + Ok(WriterInitResult { + replay_iterator, + wal_writer: Box::new(BTreeMapWalWriter::new(self.clone())), + }) + } +} + +fn range_contains(range: &WalFileRange, wal_file_id: u64) -> bool { + let starts_before = match range.0 { + Bound::Included(start) => wal_file_id >= start, + Bound::Excluded(start) => wal_file_id > start, + Bound::Unbounded => true, + }; + let ends_after = match range.1 { + Bound::Included(end) => wal_file_id <= end, + Bound::Excluded(end) => wal_file_id < end, + Bound::Unbounded => true, + }; + starts_before && ends_after +} + +#[async_trait] +impl WalGc for BTreeMapWal { + async fn collect( + &self, + referenced_ranges: Vec, + _min_age: Duration, + dry_run: bool, + ) -> Result<(), WalError> { + if !dry_run { + self.files.lock().retain(|wal_file_id, _| { + referenced_ranges + .iter() + .any(|range| range_contains(range, *wal_file_id)) + }); + } + Ok(()) + } +} + +/// Supplies a separate `BTreeMapWal` for each database path so clone administration can copy +/// the source WAL into the clone's WAL namespace. +#[derive(Clone, Default)] +struct BTreeMapWalAdmin { + wals: Arc>>, +} + +impl BTreeMapWalAdmin { + fn wal(&self, path: &Path) -> BTreeMapWal { + self.wals + .lock() + .entry(path.to_string()) + .or_default() + .clone() + } +} + +#[async_trait] +impl WalAdmin for BTreeMapWalAdmin { + fn garbage_collector(&self, path: &Path) -> Arc { + Arc::new(self.wal(path)) + } + + async fn delete_wal(&self, path: &Path, dry_run: bool) -> Result, WalError> { + let key = path.to_string(); + let exists = self.wals.lock().contains_key(&key); + if exists && !dry_run { + self.wals.lock().remove(&key); + } + Ok(exists + .then(|| format!("btree-map-wal:{key}")) + .into_iter() + .collect()) + } + + async fn is_empty( + &self, + path: &Path, + _replay_after_wal_id: u64, + _wal_id_last_seen: u64, + ) -> Result { + Ok(self.wal(path).files.lock().values().all(Vec::is_empty)) + } + + async fn clone_wal( + &self, + from_path: &Path, + from_manifest: VersionedManifest, + to_path: &Path, + ) -> Result<(u64, u64), WalError> { + let replay_after_wal_id = from_manifest.replay_after_wal_id(); + let copied = self + .wal(from_path) + .files + .lock() + .range((Bound::Excluded(replay_after_wal_id), Bound::Unbounded)) + .map(|(id, batch)| (*id, batch.clone())) + .collect::>(); + let last_wal_file_id = copied + .last_key_value() + .map(|(id, _)| *id) + .unwrap_or(replay_after_wal_id); + *self.wal(to_path).files.lock() = copied; + Ok((replay_after_wal_id, last_wal_file_id)) + } +} + +fn test_settings() -> Settings { + Settings { + flush_interval: None, + compactor_options: None, + garbage_collector_options: None, + ..Settings::default() + } +} + +async fn open_db(path: Path, object_store: Arc, wal: BTreeMapWal) -> Db { + Db::builder(path, object_store) + .with_settings(test_settings()) + .with_wal_writer(Box::new(wal)) + .build() + .await + .expect("failed to open database with custom WAL") +} + +async fn close_without_memtable_flush(db: &Db) { + db.close_with_options(CloseOptions::default().with_flush_type(None)) + .await + .expect("failed to close database") +} + +#[tokio::test] +async fn custom_wal_basic_write_and_recovery() { + let object_store: Arc = Arc::new(InMemory::new()); + let path = Path::from("/custom-wal/basic-recovery"); + let wal = BTreeMapWal::default(); + + let db = open_db(path.clone(), Arc::clone(&object_store), wal.clone()).await; + db.put(b"key-1", b"value-1").await.expect("put failed"); + db.put(b"key-2", b"value-2").await.expect("put failed"); + + assert_eq!(wal.file_ids(), vec![1, 2]); + assert!(wal.snapshot().values().all(|batch| batch.len() == 1)); + close_without_memtable_flush(&db).await; + + let recovered = open_db(path, object_store, wal).await; + assert_eq!( + recovered + .get(b"key-1") + .await + .expect("get failed") + .as_deref(), + Some(b"value-1".as_slice()) + ); + assert_eq!( + recovered + .get(b"key-2") + .await + .expect("get failed") + .as_deref(), + Some(b"value-2".as_slice()) + ); + close_without_memtable_flush(&recovered).await; +} + +#[tokio::test] +async fn custom_wal_db_reader_serves_wal_data() { + let object_store: Arc = Arc::new(InMemory::new()); + let path = Path::from("/custom-wal/db-reader"); + let wal = BTreeMapWal::default(); + let db = open_db(path.clone(), Arc::clone(&object_store), wal.clone()).await; + db.put(b"reader-key", b"reader-value") + .await + .expect("put failed"); + + let reader = DbReader::builder(path, object_store) + .with_reader_mode(DbReaderMode::FollowLatest) + .with_options(DbReaderOptions { + manifest_poll_interval: Duration::from_secs(60), + ..DbReaderOptions::default() + }) + .with_wal_reader(Arc::new(wal)) + .build() + .await + .expect("failed to open database reader"); + assert_eq!( + reader + .get(b"reader-key") + .await + .expect("reader get failed") + .as_deref(), + Some(b"reader-value".as_slice()) + ); + + reader.close().await.expect("failed to close reader"); + close_without_memtable_flush(&db).await; +} + +#[tokio::test] +async fn custom_wal_garbage_collects_unused_ranges() { + let object_store: Arc = Arc::new(InMemory::new()); + let path = Path::from("/custom-wal/gc"); + let wal = BTreeMapWal::default(); + let db = open_db(path.clone(), Arc::clone(&object_store), wal.clone()).await; + + for id in 1..=3 { + db.put( + format!("key-{id}").as_bytes(), + format!("value-{id}").as_bytes(), + ) + .await + .expect("put failed"); + } + db.flush_with_options(FlushOptions { + flush_type: FlushType::MemTable, + }) + .await + .expect("memtable flush failed"); + db.put(b"key-4", b"value-4").await.expect("put failed"); + close_without_memtable_flush(&db).await; + assert_eq!(wal.file_ids(), vec![1, 2, 3, 4]); + + let gc = GarbageCollectorBuilder::new(path, object_store) + .with_wal_gc(Arc::new(wal.clone())) + .with_options(GarbageCollectorOptions { + manifest_options: None, + wal_options: Some(GarbageCollectorDirectoryOptions { + interval: None, + min_age: Duration::ZERO, + dry_run: false, + }), + wal_fence_options: None, + compacted_options: None, + compactions_options: None, + detach_options: None, + ..GarbageCollectorOptions::default() + }) + .build(); + gc.run_gc_once().await; + + // The latest manifest references the replay boundary (3) and everything after it. + assert_eq!(wal.file_ids(), vec![3, 4]); +} + +#[tokio::test] +async fn custom_wal_clone_has_the_same_data() { + let object_store: Arc = Arc::new(InMemory::new()); + let source_path = Path::from("/custom-wal/clone-source"); + let clone_path = Path::from("/custom-wal/clone-destination"); + let wal_admin = BTreeMapWalAdmin::default(); + let source_wal = wal_admin.wal(&source_path); + + let source_db = open_db( + source_path.clone(), + Arc::clone(&object_store), + source_wal.clone(), + ) + .await; + source_db + .put(b"clone-key-1", b"clone-value-1") + .await + .expect("put failed"); + source_db + .put(b"clone-key-2", b"clone-value-2") + .await + .expect("put failed"); + close_without_memtable_flush(&source_db).await; + + Admin::builder(clone_path.clone(), Arc::clone(&object_store)) + .with_wal_admin(Arc::new(wal_admin.clone())) + .build() + .create_clone_builder_from_source(CloneSourceSpec::new(source_path)) + .build() + .await + .expect("failed to create clone"); + + let clone_wal = wal_admin.wal(&clone_path); + assert_eq!(clone_wal.snapshot(), source_wal.snapshot()); + let clone_db = open_db(clone_path, object_store, clone_wal).await; + assert_eq!( + clone_db + .get(b"clone-key-1") + .await + .expect("get failed") + .as_deref(), + Some(b"clone-value-1".as_slice()) + ); + assert_eq!( + clone_db + .get(b"clone-key-2") + .await + .expect("get failed") + .as_deref(), + Some(b"clone-value-2".as_slice()) + ); + close_without_memtable_flush(&clone_db).await; +} diff --git a/slatedb/tests/prefix_filter.rs b/slatedb/tests/prefix_filter.rs index ed7ec109de..99510a4353 100644 --- a/slatedb/tests/prefix_filter.rs +++ b/slatedb/tests/prefix_filter.rs @@ -121,7 +121,7 @@ mod composite_filters { let put = PutOptions::default(); let write = WriteOptions { await_durable: false, - seqnum: 0, + ..WriteOptions::default() }; // Write each batch in its own SST so multiple SSTs participate in the // read path and the filter has something to actually skip. @@ -301,7 +301,7 @@ mod subrange { let put = PutOptions::default(); let write = WriteOptions { await_durable: false, - seqnum: 0, + ..WriteOptions::default() }; let ssts: &[&[&[u8]]] = &[ &[b"aaa1", b"ccc1"], // sandwich @@ -495,7 +495,7 @@ mod empty_prefix_filter { let put = PutOptions::default(); let write = WriteOptions { await_durable: false, - seqnum: 0, + ..WriteOptions::default() }; for key in [b"a".as_slice(), b"b".as_slice()] { db.put_with_options(key, b"v", &put, &write) @@ -585,7 +585,7 @@ mod prop_test { let put_opts = PutOptions::default(); let write_opts = WriteOptions { await_durable: false, - seqnum: 0, + ..Default::default() }; for (i, key) in keys.iter().enumerate() { let value = format!("v{}", i).into_bytes(); diff --git a/website/public/charts/benchmark-balanced-get-latency.html b/website/public/charts/benchmark-balanced-get-latency.html new file mode 100644 index 0000000000..95daba56e2 --- /dev/null +++ b/website/public/charts/benchmark-balanced-get-latency.html @@ -0,0 +1,166 @@ + + + + + +SlateDB balanced workload get application latency + + + +

+

Application latency · get

+
+ avg + p50 + p99 + p99.9 + full benchmark result ↗ +
+
+ + SlateDB balanced workload get application latency + Average and percentile get latency over the balanced workload measurement, with prominent spikes in the p99 and p99.9 series. + +
+
+
+ + + diff --git a/website/public/charts/benchmark-sustained-ingest-put.html b/website/public/charts/benchmark-sustained-ingest-put.html new file mode 100644 index 0000000000..c6c5754500 --- /dev/null +++ b/website/public/charts/benchmark-sustained-ingest-put.html @@ -0,0 +1,138 @@ + + + + + +SlateDB sustained-ingest application throughput + + + +
+

Application throughput

+
+ put + published average (91.58 MiB/s) + full benchmark result ↗ +
+
+ + SlateDB sustained-ingest application put throughput + One-second application put throughput over roughly twenty minutes, with repeated drops as SlateDB throttles writes. The published average is 91.58 MiB per second. + +
+
+
+ + + diff --git a/website/public/charts/distributed-compaction-scaling.html b/website/public/charts/distributed-compaction-scaling.html new file mode 100644 index 0000000000..b2e506de69 --- /dev/null +++ b/website/public/charts/distributed-compaction-scaling.html @@ -0,0 +1,133 @@ + + + + + +SlateDB — distributed compaction scaling (workers vs throughput, RFC-0025) + + + + +
+

Throughput scales with workers

+
+
+
+
+ + + + + + + diff --git a/website/public/charts/subcompaction-scaling.html b/website/public/charts/subcompaction-scaling.html new file mode 100644 index 0000000000..44af0cc5de --- /dev/null +++ b/website/public/charts/subcompaction-scaling.html @@ -0,0 +1,169 @@ + + + + + +SlateDB — subcompaction scaling (16 GB compaction, RFC-0028) + + + + +
+

Near-linear speedup

+
+
+ +

Wall-clock compaction time

+
+
+
+
+ + + + + + + diff --git a/website/public/img/logos/prisma.svg b/website/public/img/logos/prisma.svg new file mode 100644 index 0000000000..0f42a21023 --- /dev/null +++ b/website/public/img/logos/prisma.svg @@ -0,0 +1,3 @@ + + + diff --git a/website/src/components/BenchmarkComparisonTable.astro b/website/src/components/BenchmarkComparisonTable.astro new file mode 100644 index 0000000000..7ed5cf15a4 --- /dev/null +++ b/website/src/components/BenchmarkComparisonTable.astro @@ -0,0 +1,237 @@ +--- +type Support = boolean | 'manual'; + +const benchmarks = ['YCSB', 'KVBench', 'db_bench', 'Tectonic']; +const groups: Array<{ + label: string; + rows: Array<{ capability: string; support: Support[] }>; +}> = [ + { + label: 'Operations', + rows: [ + { capability: 'Insert', support: [true, true, true, true] }, + { capability: 'Update', support: [true, true, true, true] }, + { capability: 'Read-modify-write', support: [true, false, true, true] }, + { capability: 'Point query', support: [true, true, true, true] }, + { capability: 'Empty point query', support: [false, true, true, true] }, + { capability: 'Range query', support: [true, true, true, true] }, + { capability: 'Point delete', support: [false, true, true, true] }, + { capability: 'Empty point delete', support: [false, true, false, true] }, + { capability: 'Range delete', support: [false, true, true, true] }, + ], + }, + { + label: 'Distributions', + rows: [ + { capability: 'Uniform', support: [true, true, true, true] }, + { capability: 'Normal', support: [false, true, true, true] }, + { capability: 'Beta', support: [false, true, false, true] }, + { capability: 'Zipfian', support: [true, true, false, true] }, + { capability: 'Exponential', support: [false, false, true, true] }, + { capability: 'Log normal', support: [false, false, false, true] }, + { capability: 'Poisson', support: [false, false, false, true] }, + { capability: 'Weibull', support: [false, false, false, true] }, + { capability: 'Pareto', support: [false, false, true, true] }, + ], + }, + { + label: 'Properties', + rows: [ + { capability: 'Dynamic workload shifts', support: [false, 'manual', 'manual', true] }, + { capability: 'Context-aware shifting', support: [false, false, false, true] }, + { capability: 'Data sortedness', support: [false, false, false, true] }, + { capability: 'Variable query selectivity', support: [true, false, true, true] }, + { capability: 'Variable key-value length', support: [false, false, false, true] }, + { capability: 'Temporality-based access', support: [false, false, true, true] }, + { capability: 'Customizable key prefix', support: [false, false, 'manual', true] }, + { capability: 'Composite keys', support: [false, false, false, true] }, + ], + }, +]; +--- + +
+
+ + + + + + + + + + + + {benchmarks.map((benchmark) => ( + + ))} + + + {groups.map((group) => ( + + {group.rows.map((row, rowIndex) => ( + + {rowIndex === 0 && ( + + )} + + {row.support.map((value) => ( + + ))} + + ))} + + ))} +
Feature coverage across YCSB, KVBench, db_bench, and Tectonic
CategoryCapability{benchmark}
+ {group.label} + {row.capability} + {value === true ? ( + <>Supported + ) : value === 'manual' ? ( + <>Requires significant manual intervention + ) : ( + Not supported + )} +
+
+
* Requires significant manual intervention.
+
+ + diff --git a/website/src/components/PostToc.astro b/website/src/components/PostToc.astro index b089a484ef..8b0f661df7 100644 --- a/website/src/components/PostToc.astro +++ b/website/src/components/PostToc.astro @@ -6,7 +6,7 @@ interface Props { } const { headings } = Astro.props; -const items = headings.filter((h) => h.depth >= 2 && h.depth <= 4); +const items = headings.filter((h) => h.depth >= 2 && h.depth <= 3); --- {items.length > 0 && ( diff --git a/website/src/content.config.ts b/website/src/content.config.ts index c23728a710..af37251b23 100644 --- a/website/src/content.config.ts +++ b/website/src/content.config.ts @@ -17,6 +17,9 @@ export const collections = { authorGithub: z.string().optional(), // Optional per-post social image; falls back to the site default. ogImage: z.string().optional(), + // Optional canonical URL for cross-posted/syndicated content; points + // search engines at the original. Falls back to this page's own URL. + canonicalUrl: z.string().url().optional(), }), }), }; diff --git a/website/src/content/blog/benchmarking-slatedb.mdx b/website/src/content/blog/benchmarking-slatedb.mdx new file mode 100644 index 0000000000..0655c21143 --- /dev/null +++ b/website/src/content/blog/benchmarking-slatedb.mdx @@ -0,0 +1,86 @@ +--- +title: "Benchmarking SlateDB" +pubDate: 2026-08-12 +author: Chris +authorGithub: criccomini +# ogImage: /img/some-custom-card.jpg # optional per-post override +--- + +import ChartEmbed from '../../components/ChartEmbed.astro'; +import BenchmarkComparisonTable from '../../components/BenchmarkComparisonTable.astro'; + +SlateDB has been around for a few years now. We've begun to pay more attention to performance lately. As [Kent Beck](https://en.wikipedia.org/wiki/Kent_Beck) (purportedly) said, ["Make It Work, Make It Right, Make It Fast."](https://wiki.c2.com/?MakeItWorkMakeItRightMakeItFast) We feel we've earned the right to make SlateDB fast. Of course, before we improve performance, we must measure it. That means benchmarks. + +Benchmarks are tricky. It's easy to do something that is perceived as marketing, hyperbole, or outright dishonesty. It's questionable whether public benchmarks are even useful anymore. A frontier LLM is capable of building excellent bespoke benchmarks for each user. + +Yet we found ourselves rebuilding SlateDB's benchmark suite a few weeks back. We--the SlateDB developers--still need them to measure performance. We also believe the data will help users gauge whether SlateDB meets their performance needs. + +This post covers the benchmarking philosophy we landed on, how we benchmarked SlateDB, and some things we learned along the way. + +(You can head over to [benchmark.slatedb.io](https://benchmark.slatedb.io) if you just want the results.) + +## Legacy benchmarks + +We have had benchmarks for quite a while: both [Criterion microbenchmarks](https://github.com/bheisler/criterion.rs) and a [`benchmark-db.sh`](https://github.com/slatedb/slatedb/blob/f88be86d17ac53260d3684edbc8f82811d945b5c/slatedb-bencher/benchmark-db.sh) script. We also run a nightly `benchmark-db.sh` job against a [Tigris](https://www.tigrisdata.com/) object store bucket. (A special thanks to them for donating the bucket to us free of charge.) + +These benchmarks were largely ignored. The output was buried in a [GitHub Actions](https://docs.github.com/actions) summary. It didn't alert when significant regressions occurred, either. If you managed to find it, the output was hard to read. + +Our goal was to make SlateDB's benchmarks easier to find and understand. We also wanted better diagnostic information so developers could visualize SlateDB's behavior under different workloads. + +## Benchmark suite + +We drew inspiration from [ClickBench](https://github.com/ClickHouse/ClickBench), [DuckDB’s benchmark suite](https://duckdb.org/docs/current/dev/benchmark), Lucene’s [nightly benchmarks](https://lucene.apache.org/core/developer.html), and Apache Iggy’s [public benchmark dashboard](https://iggy.apache.org/blogs/2025/02/17/transparent-benchmarks/). [Tectonic: Bridging Synthetic and Real-World Workloads for Key-Value Benchmarking](https://scholarworks.brandeis.edu/esploro/outputs/conferencePresentation/Tectonic-Bridging-Synthetic-and-Real-World-Workloads/9924594037601921) provides a good overview of the space, for those looking to get up to speed. + +Table 1 in the Tectonic paper is particularly relevant. It compares [YCSB](https://github.com/brianfrankcooper/YCSB), [KVBench](https://dl.acm.org/doi/abs/10.1145/3662165.3662765), RocksDB’s [`db_bench`](https://github.com/facebook/rocksdb/wiki/RocksDB-Overview), and Tectonic's own workloads. + + + +Many of the YCSB and `db_bench` workloads cover the behavior users are likely to care about. We chose to adopt a subset of those. If a user needs a more specialized workload, they should run it themselves. The [benchmark repository](https://github.com/slatedb/slatedb-benchmark/) is available for that purpose. + +We also added cost metrics. Throughput and latency matter, but SlateDB exists in part because object storage changes the economics of persistence. Databases should be cheap to run. If it is not, we need to know. + +The suite is made up of three components: + +- A standalone benchmark repository with the suite and its configuration. +- [GitHub Actions](https://github.com/slatedb/slatedb-benchmark/actions) workflows that run it. +- A public [benchmark website](https://benchmark.slatedb.io/) where the results are easier to inspect. + +## Methodology + +We follow YCSB and `db_bench` workloads where possible, but SlateDB does not map perfectly to either. SlateDB is designed around object storage. A cache miss is far more costly, both in terms of money and latency. Misses incur a metered object store API call and a remote network hop. Rather than adjust our configurations to accommodate this, we opted to keep configuration close to SlateDB's defaults. + +The workloads run on [AWS Graviton](https://aws.amazon.com/ec2/graviton/) machines managed by [WarpBuild](https://warpbuild.com) in `us-east-1`. They talk to an [Amazon S3](https://aws.amazon.com/s3/) bucket in the same region and to [Tigris](https://www.tigrisdata.com/) through its `us-east-1` on-ramp. We use a relatively large machine instance ([m8g.2xlarge](https://aws.amazon.com/ec2/instance-types/m8g/)) because compaction is CPU-intensive. + +We configured SlateDB’s cache to hold roughly 10% of the database. The rest lives in object storage. A `db_bench` run of RocksDB keeps the entire database on local disk. This is clearly an apples to oranges comparison. These settings are closer to the way many users configure SlateDB, though. + +Another tradeoff applies to mixed read/write workloads. A read that misses the cache can take tens or hundreds of milliseconds while data comes back from object storage. If one task does both reads and writes, those misses hold back its write throughput. We could separate readers and writers or add more parallelism to report a stronger aggregate throughput number. We kept the configuration closer to RocksDB's workload instead. + +## Findings + +Our first discovery was that routing over the public internet is unpredictable. At one point, we ran in [Hetzner](https://www.hetzner.com/) and expected a nearby path to our Tigris object store bucket. The traffic took a longer route through a different on-ramp, which showed up in latency. We also found local routing behavior that caused tail-latency problems in `us-east-1`. This led to very sawtoothed ingestion charts as SlateDB throttled writes. We eventually worked through these with our cloud provider. + + + +The ingestion benchmark also pushed us toward a simple form of trivial moves. When ingested data has no overlapping key ranges, SlateDB can move it directly from L0 into the next sorted run without rewriting it. RocksDB [supports this](https://github.com/facebook/rocksdb/wiki/Leveled-Compaction#trivial-move), but we hadn't bothered to implement it. Our `sustained-ingest` workload showed it was worth doing. We've opted to disable the setting in the benchmark so we continue to represent baseline performance. We have [a more sophisticated trivial move](https://github.com/slatedb/slatedb/issues/1110) implementation in the works, too. + +Another result came from our new cost performance metrics. We recently added [distributed compaction](https://slatedb.io/blog/compaction-roadmap) support. Compaction can now run on one or more remote machines. This implementation involves periodically checking persisted compaction state in object storage to see if new work is scheduled. + +Most SlateDB deployments keep the writer, compactor, and garbage collector in one process, though. We were still using remote-style object store polling in that scenario. The benchmark showed that an idle database with default settings cost roughly $40 per-month in API polling fees. We made the local coordination path in-memory while preserving the state needed for remote compactors to participate. The idle cost fell to under $5. In practice, you would tune your compactor settings or close idle databases, but $40 per-month is clearly excessive. + +The benchmarks also changed how we think about disk caching. SlateDB has a block cache and an object-store cache. The block cache holds pieces such as data blocks, index blocks, filters, and metadata. The object-store cache works at a larger granularity; its default partition size is 4 MiB. That granularity is expensive when a cache miss occurs. A request for a small key can trigger a 4 MiB fetch from object storage. This appeared as significant latency spikes in our P99 metrics. + + + +We are now planning a [more capable object-store mirror cache](https://github.com/slatedb/slatedb/issues/1980) with read-through, write-through, and write-back modes. The division of labor is clearer, too. If you want to keep the whole database local, an object-level cache makes sense. If you only want partial local caching, the [Foyer hybrid cache](https://github.com/foyer-rs/foyer) is usually a better fit. It already supports eviction and works with much smaller units of data so a cache miss is less costly. [ZeroFS](http://zerofs.net/) has shown us that a [prefetching cache](https://github.com/Barre/ZeroFS/blob/main/zerofs/src/object_store_prefetch.rs) for spatial locality is also possible. + +## More to come + +Performance tuning is a never-ending endeavor. We will continue driving SlateDB's cost and latency down while improving its throughput. We now have a means to evaluate that progress. The results are available at [benchmark.slatedb.io](https://benchmark.slatedb.io/) if we've piqued your curiosity. diff --git a/website/src/content/blog/compaction-roadmap.mdx b/website/src/content/blog/compaction-roadmap.mdx new file mode 100644 index 0000000000..882c8ff431 --- /dev/null +++ b/website/src/content/blog/compaction-roadmap.mdx @@ -0,0 +1,437 @@ +--- +title: The Past, Present, and Future of Compaction in SlateDB +pubDate: 2026-07-13 +author: Ryan Dielhenn, Almog Gavra +canonicalUrl: https://ryandielhenn.github.io/blog/distributed-compaction/ +--- + +import ChartEmbed from '../../components/ChartEmbed.astro'; + +[SlateDB](https://slatedb.io) is an embedded key-value store built on object +storage. + +SlateDB uses a Log-Structured Merge Tree, or LSM for short, to batch writes to +object storage. Incoming writes land in an in-memory buffer called the +memtable. Once the memtable fills up, it is flushed to an immutable file stored +in object storage called a Sorted String Table (SST). + +LSM trees are structured in multiple levels, and the first time an SST lands +in object store it is put in the first level known as L0. From there, SlateDB +uses a process called compaction, where SSTs are grouped and merged together as +each tier fills up. + +LSM trees let you tune the tradeoffs between read, write, and space +amplification. I recommend reading [this +blog](https://www.bitsxpages.com/p/understanding-lsm-trees-via-read) by Almog +Gavra if you want to know more about these tradeoffs. + +The following is what you would see if you listed the contents of an object +storage bucket path used by SlateDB: + +```text +manifest/ + 00000000001.manifest # This is a snapshot of the database state: + 00000000002.manifest # SST lists, watermarks, epochs, external dbs, checkpoints etc. + 00000000003.manifest + ... +compactions/ + 00000000001.compactions # Jobs scheduled for compaction by the + 00000000002.compactions # compaction coordinator. + 00000000003.compactions + ... +compacted/ + .sst # This is compacted data + .sst # L0 ssts also happen to live here which have not been compacted yet + .sst + ... +wal/ # This is the write ahead log. Writes land here first so that they can be replayed in a failure scenario. + 00000000001.sst + 00000000002.sst + 00000000003.sst +gc/ + manifest.boundary # Garbage collector deletes .manifest versions at or below this Boundary + compactions.boundary # Garbage collector deletes .compactions files at or below this Boundary +``` + +All of this is hidden from the user under simple put/get/scan APIs. + +## Compaction + +Compaction is a critical background process of the LSM tree that takes Sorted +String Tables (SST for short) and merges them to produce an output SST with +non-repeating keys. + +This process does a few things. When multiple SSTs share keys, merging them +removes duplicate entries and cleans up tombstones left behind by deletes, +reducing space amplification. It also reduces the number of SSTs that need to +be read to find a key by improving the locality of sorted data into longer +runs: + +```ascii-art +┌SST A · newest────────────┐ ┌SST B─────────────────────┐ ┌SST C · oldest────────────┐ +│orders:1042 @14 shipped │ │orders:1042 @8 packed │ │orders:1042 @3 placed │ +│orders:1058 @13 paid │ │orders:1058 @7 pending │ │orders:1023 @2 paid │ +│orders:1023 @12 ⌫ deleted│ │orders:1023 @6 shipped │ │users:42 @1 a@old.co │ +│users:42 @11 a@new.co │ │users:91 @5 plan:pro │ │users:91 @0 plan:free│ +└──────────────────────────┘ └──────────────────────────┘ └──────────────────────────┘ + + │ │ │ + └──────────────────────────────┼──────────────────────────────┘ + ▼ + ┌k-way merge · min-heap────────────────────┐ + │pop smallest (key, -LSN) across streams │ + └──────────────────────────────────────────┘ + ▼ + ┌per-key winner · highest LSN wins───────────────────┐ + │orders:1042 keep @14 shipped · drop @8, @3 │ + │orders:1023 ⌫ tombstone @12 · drop @6, @2 │ + │users:42 keep @11 a@new.co · drop @1 │ + │orders:1058 keep @13 paid · drop @7 │ + │users:91 keep @5 plan:pro · drop @0 │ + └────────────────────────────────────────────────────┘ + ▼ + ┌output SST · one current row per key────────────────┐ + │orders:1042 → shipped users:42 → a@new.co │ + │orders:1058 → paid users:91 → plan:pro │ + │orders:1023 → ⌫ dropped (tombstone removed) │ + └────────────────────────────────────────────────────┘ +``` + +

The LSM compaction merge step: a k-way merge keeps each +key's highest-LSN version and drops the rest. Rows are written key @LSN → +value; ⌫ marks a tombstone (a deleted key).

+ +## Distributed Compaction + +A single compactor is a bottleneck: if it cannot keep pace with write +throughput, the whole system degrades in two stages: + +1. as uncompacted SSTs pile up in L0, more files need to be scanned to + find a key, increasing read latency. +2. once the L0 file count reaches `l0_max_ssts`, the flusher stops writing + immutable memtables to L0. Those memtables accumulate in memory until + `max_unflushed_bytes` is exceeded, at which point SlateDB applies + backpressure that stalls writes from being durably written to object + storage. + +A lagging compactor therefore degrades read latency first, then write throughput. + +In 0.14.1 we introduced distributed compaction, which mitigates this +issue by allowing SlateDB to leverage multiple concurrent workers +to run compactions. + +```ascii-art + Single compactor │ Distributed workers + ┌write path · flushes fast─────────┐ │ ┌write path · flushes fast─────────┐ + └──────────────────────────────────┘ │ └──────────────────────────────────┘ + ▼ │ ▼ + ┌L0 SST count rising───────────────┐ │ ┌compaction keeps up───────────────┐ + │█ █ █ █ █ █ █ █ █ █ █ █ │ │ │█ █ █ ░ ░ ░ ░ ░ ░ ░ ░ ░ │ + │read latency degrades │ │ │drained as fast as it fills │ + └──────────────────────────────────┘ │ └──────────────────────────────────┘ + ▼ │ ▼ + ┌flusher stops at l0_max_ssts──────┐ │ ┌.compactions · job queue──────────┐ + │immutable memtables pile in RAM │ │ └──────────────────────────────────┘ + └──────────────────────────────────┘ │ ┌────────────┬────────────┐ + ▼ │ ▼ ▼ ▼ + ┌max_unflushed_bytes exceeded──────┐ │ ┌worker 1──┐ ┌worker 2──┐ ┌worker N──┐ + │back-pressure stalls writes │ │ │job a, b │ │job c, d │ │job e, f │ + └──────────────────────────────────┘ │ └──────────┘ └──────────┘ └──────────┘ + ▼ │ │ │ │ + ┌1 compactor draining──────────────┐ │ └────────────┼────────────┘ + │capped by one node's CPU + net │ │ ▼ merged SSTs + └──────────────────────────────────┘ │ ┌sorted runs───────────────────────┐ + │ │█ █ █ █ █ █ █ █ │ +A lagging compactor degrades the │ └──────────────────────────────────┘ +whole system: reads slow first, │ +then writes stall. │ Add workers to raise the ceiling; + │ the single-writer manifest + │ invariant is preserved. +``` + +

How distributing compaction across stateless workers relieves the single-compactor bottleneck.

+ +Distributed compaction works by designating a single coordinator for compactions +that handles scheduling. This coordinator writes the results of scheduling into +a shared [transactional +object](https://github.com/slatedb/slatedb/blob/main/slatedb-txn-obj/README.md) +on object storage called the `.compactions` file, and workers poll this file to +claim compactions, updating that file when they have made progress on executing +a transaction. + +To understand its effect on scaling we ran a benchmark that clears out a backlog +of independent per-segment ~2GB compactions while varying the number of +workers, each pinned to a single vCPU. Since the jobs are independent and each +runs on its own worker, throughput scales near-linearly all the way to six +workers: 5.2× the single-worker rate while holding above 85% parallel +efficiency. + + + +In addition to allowing for improved compaction throughput, distributed compaction +enables some nice quality of life operational improvements: + +1. You can migrate from running compaction embedded on the writer to running it + standalone (with one or more workers) without fencing the active writer. +2. You can start/stop compactor workers without worrying about fencing at all. +3. You can run compactors on a compute framework like kubernetes without + installing a separate controller. + +The full design of distributed compactions is in +[RFC-0025](/rfcs/0025-distributed-compaction) and is well worth a read. + +## Future benefits and work + +The performance and operational improvements of distributed compaction are +just the start. The door is wide open for future enhancements that take +advantage of these stateless compaction workers. Below are just a few examples +of extensions made possible by the stateless workers added in RFC-0025. + +### Concurrent L0 Compactions + +Ideally we want to parallelize compaction work, but it helps to separate two +axes of parallelism that are easy to conflate. + +The first axis is running *independent* compactions at the same time. There are +two different types of independent compactions: + +1. Disjoint sorted run compactions. These are compactions that target already + compacted data from different levels of the tree (e.g. `L0→SR1` and + `[SR2,SR3]→SR4`). +2. Compactions from different segments. + [RFC-0024](/rfcs/0024-segment-oriented-compaction) + introduced segmented compaction which allows SlateDB to maintain multiple + LSM trees that share a single WAL/memtable. Compactions that target + different segments are independent, even targeting L0. + +Distributed compaction is what lets you execute these across more than one +machine: a single embedded worker is capped by one machine's CPU and I/O, and +`max_concurrent_compactions` only stretches that one machine so far. Spreading +independent jobs across a pool of workers is the bottleneck relief this post is +about. + +The second axis is parallelizing L0 compactions within the same segment, and +here it's worth not overselling: distributed compaction does **not** unlock it +on its own. + +`last_compacted_l0_sst_view_id` is a single cursor over a segment's L0 list, +and "already compacted" means "at or below the cursor." A single monotonic +boundary can't represent two disjoint, in-flight L0 consumptions at once, so L0 +compaction within a tree is serialized whether that tree is served by one +embedded worker or a fleet of remote ones. RFC-24 calls this out directly: + +> Parallel L0 compaction within a single segment is a separate concern tied to +> the watermark's single-cursor design and is not addressed here. + +```ascii-art +segment L0 list · oldest ──────────────────▶ newest +┌────┐ ┌────┐ ┌────┐ ┌────┐ ┌────┐ ┌────┐ +│ L1 │ │ L2 │ │ L3 │ │ L4 │ │ L5 │ │ L6 │ +└────┘ └────┘ └────┘ └────┘ └────┘ └────┘ + ▲ + cursor = last_compacted_l0_sst_view_id + (at/below = consumed · above = live) + +┌Job A · consume {L3, L4}────┐ ┌Job B · consume {L5, L6}────┐ +│still running … │ │finishes first │ +└────────────────────────────┘ └────────────────────────────┘ + +One monotonic cursor can't hold two in-flight consumptions: B commits, +the cursor jumps past L6, and L3/L4 are GC'd while A is still running. +``` + +

A single monotonic cursor can't represent two disjoint, in-flight L0 consumptions in the same segment.

+ +That note is just as true after RFC-25. Moving work onto stateless workers +changes where a compaction runs, not whether one L0 compaction can be split +in two. + +Closing that gap takes one of two things (or both), and neither is distributed +compaction: + +#### Improvement 1: Subcompactions + +Subcompactions ([RFC-0028](/rfcs/0028-subcompactions)). +Rather than splitting the L0 list across multiple compactions (which the +watermark forbids), a subcompaction keeps it as one logical compaction and +splits the *key range* into sub-ranges that run in parallel. The parent +commits a single manifest update that advances the watermark exactly once over +the whole consumed set, so the single-cursor invariant is never violated which +sidesteps the problem instead of fighting it. + +Subcompactions parallelize across cores on one worker, which composes +cleanly with distributed compaction parallelizing across workers; allowing +separate workers to claim subcompactions of the same parent compaction is +explicitly labeled as future work in RFC-0028 and out of scope. + +```ascii-art +┌one logical compaction · live set {L3 … L6}─┐ +│┌────┐ ┌────┐ ┌────┐ ┌────┐ │ +││ L3 │ │ L4 │ │ L5 │ │ L6 │ │ +│└────┘ └────┘ └────┘ └────┘ │ +└────────────────────────────────────────────┘ + split by key range + ▼ ▼ +┌sub-range A · core 1┐ ┌sub-range B · core 2┐ +│keys (−∞, k) │ │keys [k, +∞) │ +│output SST(s) │ │output SST(s) │ +└────────────────────┘ └────────────────────┘ + │ │ + └───────────┬───────────┘ + ▼ +┌single manifest commit──────────────────────┐ +│advance cursor once: before L3 → after L6 │ +└────────────────────────────────────────────┘ +``` + +

Parallelism from splitting one compaction by key range; the cursor still advances exactly once per parent compaction.

+ +In practice, splitting a single 16 GB compaction into key-range subcompactions +scales almost linearly. The results below are from a benchmark running a 12-way +split ~11× faster than the baseline no-subcompactions run, collapsing a 2m37s +compaction to ~14s while holding above 90% parallel efficiency. + + + +

Benchmark evidence for subcompactions (RFC-0028): measured speedup tracks the ideal-linear reference, holding above 90% efficiency through a 12-way split.

+ +#### Improvement 2: Reworking the L0 Watermark + +Reworking the watermark to track a *set* of consumed L0 SSTs instead of a +single cursor. This is the more invasive change, but it's what would let two L0 +parent compactions in the same segment advance independently. + +```ascii-art +segment L0 list · oldest ──────────────────▶ newest +track a SET of consumed SSTs ( ● consumed ○ live ) +┌────┐ ┌────┐ ┌────┐ ┌────┐ ┌────┐ ┌────┐ +│L1 ●│ │L2 ●│ │L3 ○│ │L4 ○│ │L5 ○│ │L6 ○│ +└────┘ └────┘ └────┘ └────┘ └────┘ └────┘ + ▲ ▲ ▲ ▲ + └───┬───┘ └───┬───┘ + Job A Job B + {L3,L4} {L5,L6} + independent independent + +A set-valued watermark lets two L0 compactions in the same segment +advance independently so neither GC's the other's live inputs. +``` + +

Track a set of compacted SSTs instead of one boundary, so two L0 compactions in the same segment advance independently.

+ +To summarize, distributed compaction removes the restriction that compaction +runs on a single node, subcompactions remove the restriction that a single +compaction runs on a single CPU, and the watermark design (still to be +implemented as of July 2026) will allow us to run concurrent L0 compactions +within a single segment. + +### Priority Based Compaction Routing + +Once compactions are jobs claimed by stateless workers rather than steps run +inline by one process, the coordinator is free to decide *which* job goes +*where*. + +Compactions could be routed to specific workers, or pools of them, +based on priority: an L0 compaction on a segment approaching `l0_max_ssts` is +far more urgent than a routine sorted-run merge, because the former is what +stands between the database and write-stalling backpressure. High-priority jobs +could be steered to a dedicated set of low-latency workers while bulk +sorted-run merges run on cheaper, best-effort capacity. + +This will turn the worker pool into a scheduling surface where compaction +resources follow the work that is most likely to degrade read and write +latency. + +```ascii-art +┌.compactions · job queue with priority hints────────┐ +│L0 drain · hot large-tier merge · heavy │ +└────────────────────────────────────────────────────┘ + │ │ +┌router · reads priority, dispatches to matching pool┐ +└────────────────────────────────────────────────────┘ + ▼ ▼ +┌hot pool────────────────┐ ┌heavy pool──────────────┐ +│L0 drains · latency │ │big merges · thruput │ +│████ ████ ████ ████ │ │██████████ ██████████ │ +└────────────────────────┘ └────────────────────────┘ +``` + +

Priority hints let the coordinator route hot L0 drains and heavy sorted-run merges to matching worker pools.

+ +### Shared Worker Pools + +Because the workers are stateless, nothing about a compaction job ties it to a +single database instance. + +A shared worker pool serving many instances would significantly reduce the +I/O-bound threads each instance has to reserve for itself, and let instances +trade compaction resources as needed e.g. an idle database contributes its +share of the pool to a neighbor that is busy ingesting. Without it, capacity is +sized per database for that database's worst case. This is true of an embedded +compactor and equally of a remote fleet dedicated to one instance, unless you +build per-DB autoscaling. + +Pooling that capacity absorbs those bursts and raises overall utilization, +while priority-based routing decides how the shared pool is divided when +several instances contend for it at once. + +```ascii-art + Per-DB workers │ Shared workers +┌DB α────────┐ ┌DB β────────┐ ┌DB γ────────┐ │ ┌DB α────────┐ ┌DB β────────┐ ┌DB γ────────┐ +│ idle │ │ hot │ │ idle │ │ │ idle │ │ hot │ │ idle │ +└────────────┘ └────────────┘ └────────────┘ │ └────────────┘ └────────────┘ └────────────┘ + ▼ ▼ ▼ │ ▼ ▼ ▼ +┌pool α──────┐ ┌pool β──────┐ ┌pool γ──────┐ │ ┌.compactions┐ ┌.compactions┐ ┌.compactions┐ +│ ░ ░ ░ │ │ █ █ █ │ │ ░ ░ ░ │ │ │ α (empty) │ │ β (6 jobs) │ │ γ (1 job) │ +└────────────┘ └────────────┘ └────────────┘ │ └────────────┘ └────────────┘ └────────────┘ + │ └──────────────┼──────────────┘ +each DB reserves its own threads; │ ▼ +α and γ sit idle while β is saturated. │ ┌shared worker pool────────────────────────┐ + │ │polls every DB's .compactions │ +resources are locked per database — │ │ │ +β backs up but can't borrow α or γ. │ │ ┌───┬───┬───┬───┬───┬───┐ │ + │ │ │ β │ β │ β │ β │ γ │ ░ │ │ +you provision (and pay for) peak × N. │ │ └───┴───┴───┴───┴───┴───┘ │ + │ │ │ + │ │box = worker · β/γ = its DB · ░ = idle │ + │ └──────────────────────────────────────────┘ + │ + │ capacity follows demand: + │ β gets 4 · γ gets 1 · α gets 0 · 1 spare. + │ + │ pay for aggregate peak, not N × peak — + │ the spare flows wherever it is needed. +``` + +

A shared worker pool serving multiple SlateDB instances: capacity follows demand instead of being pinned per database.

+ +## Conclusion + +Distributed compaction (RFC-0025) removes the single-*process* ceiling on +compaction and lays down a stateless-worker foundation: jobs are claimed and +executed by workers that hold no durable state of their own. + +On its own that relieves the single-compactor bottleneck that degrades read +latency and then write throughput while improving the operational quality of +life of compaction in SlateDB. + +Just as importantly, the stateless-worker model is what makes the future work +above tractable. Reworking the watermark design, adding priority-based routing +and shared worker pool support are natural extensions now that a compaction is +just a job that any worker can claim. + +:::note[attribution] +This post was originally published on Ryan Dielhenn's blog at +[ryandielhenn.github.io](https://ryandielhenn.github.io/blog/distributed-compaction/), +and reproduced here with permission. This version has modified some text and added +additional benchmarking results. +::: + diff --git a/website/src/content/docs/docs/design/change-data-capture.mdx b/website/src/content/docs/docs/design/change-data-capture.mdx index a569ca3589..c534649b4b 100644 --- a/website/src/content/docs/docs/design/change-data-capture.mdx +++ b/website/src/content/docs/docs/design/change-data-capture.mdx @@ -1,118 +1,102 @@ --- title: Change Data Capture -description: Implement CDC by streaming write-ahead log (WAL) SSTs with WalReader +description: Stream SlateDB WAL files with a live SlateDbWalReader iterator --- import { Code } from '@astrojs/starlight/components'; import cdcExample from '/../examples/src/change_data_capture.rs?raw'; -Change data capture (CDC) turns writes in SlateDB into a durable change stream that external systems can consume. SlateDB exposes CDC through the `WalReader`, which reads write-ahead log (WAL) SST files from object storage and returns the same row entries that were written by the database. +Change data capture (CDC) turns writes in SlateDB into a durable change stream that external systems can consume. SlateDB exposes native-WAL CDC through **slatedb::wal::SlateDbWalReader**. -This page explains how to build a CDC pipeline with `WalReader`, including ordering, durability, deletes, and resume semantics. +The reader opens a live iterator from the first unconsumed WAL file ID. The iterator owns polling and waits internally when it reaches the current WAL tail; the caller owns durable cursor storage. ## When to use WAL-based CDC Use WAL-based CDC when you need: - A low-latency stream of inserts, updates, and deletes. -- A simple, append-only feed that can be replayed. -- Integration with downstream systems like Kafka, Pulsar, or data lakes. +- A durable append-only feed that can be replayed. +- Integration with downstream systems such as Kafka, Pulsar, or data lakes. -WAL-based CDC is not a bootstrap or backfill API. If you need an initial full copy of data, combine a backfill with WAL tailing as described below. +WAL-based CDC is not a bootstrap or backfill API. Combine a database backfill with WAL streaming when an initial full copy is required. -## What WalReader reads +## Streaming model -SlateDB stores WAL SSTs under the `wal/` directory in the object store. Each WAL file has a monotonically increasing ID and is written in insertion order by sequence number. `WalReader` lists these files in ID order and exposes a `WalFile` for each file in object storage. +Keep a durable cursor containing the last fully consumed WAL file ID. Use zero when starting from the beginning unless you have a more appropriate manifest-derived cursor. -Each `WalFile` exposes an `iterator()` that yields `RowEntry` values in order. The iterator is optional because the underlying object might have been deleted between listing and reading (for example, due to GC). +To consume the stream: -Each row returned by the iterator is a `RowEntry` with: +1. Compute **cursor + 1**, checking for overflow, and create one iterator with an unbounded end range beginning at that WAL file ID. +2. Repeatedly call **next()**. When the iterator reaches the current tail, the call waits and polls internally for the next WAL file. +3. Emit every row in each **WalRows** batch. +4. After the whole batch succeeds, persist its **last_consumed_wal_file_id** as the new cursor. +5. After a restart, create a new unbounded iterator beginning at the persisted cursor plus one. -- `key`: the key bytes -- `value`: a `ValueDeletable` variant (`Value`, `Merge`, or `Tombstone`) -- `seq`: a globally increasing sequence number -- `create_ts` and `expire_ts`: optional timestamps +The live iterator does not report exhaustion at the current tail; **next()** remains pending until a new WAL file is visible or an error occurs. **last_wal_file_id** remains available when an application needs a point-in-time tail snapshot, but it is not part of the streaming loop. -Deletes appear as `Tombstone`s. If your database uses merge operators, you will see `Merge` values and must apply the merge logic yourself if your sink requires materialized values. +## Batches and cursor progress -:::note +One **WalRows** batch represents one fully consumed WAL file and contains: -The database must have a write-ahead log in order to use WAL-based CDC. The write-ahead log is controlled with `Settings::wal_enabled`. `wal_enabled` defaults to `true`, and is feature-gated behind a `wal_disable` feature in Rust. Databases have a WAL unless you have deliberaly disabled it. +- **rows**: the file's row entries, in sequence order +- **last_consumed_wal_file_id**: the durable cursor after that file -::: +Empty fence WAL files produce a batch with no rows. The cursor still advances, so consumers must process the batch rather than flattening batches into rows and losing progress information. -## Visibility and durability - -`WalReader` only sees WAL SSTs that have been flushed to object storage. WAL flushes occur when: - -- The WAL buffer reaches its size threshold ([`l0_sst_size_bytes`](https://docs.rs/slatedb/latest/slatedb/config/struct.Settings.html#structfield.l0_sst_size_bytes) is used to limit WAL SST file sizes as well). -- The configured [`flush_interval`](https://docs.rs/slatedb/latest/slatedb/config/struct.Settings.html#structfield.flush_interval) elapses. -- You explicitly call a WAL or MemTable flush with [`Db::flush_with_options`](https://docs.rs/slatedb/latest/slatedb/struct.Db.html#method.flush_with_options). - -Note that a memtable flush also forces any pending WAL data to be flushed first, which guarantees the WAL is durable before L0 data is written. - -## CDC architecture - -```mermaid -flowchart LR - A[Writer] --> B[WAL buffer] - B --> C[WAL SSTs in object store] - C --> D[WalReader] - D --> E[CDC sink] -``` +Each row is a **RowEntry** with: -`WalReader` does not interpret or compact data. It simply provides a durable, ordered feed of row-level changes. +- **key**: key bytes +- **value**: a **ValueDeletable** value, merge operand, or tombstone +- **seq**: the globally increasing sequence number +- **create_ts** and **expire_ts**: optional timestamps -## Basic tailer loop +Deletes appear as tombstones. Merge operands remain raw merge values; downstream consumers must apply the merge logic themselves when materialized values are required. -This example tails all WAL files, emits row entries, and records a cursor so the stream can resume after restarts. +## Visibility and durability - +SlateDbWalReader only sees WAL SSTs flushed to object storage. WAL flushes occur when: -This pattern is safe and idempotent as long as you persist the cursor after each emitted row and skip any rows with `seq <= last_seq` when resuming. +- The WAL buffer reaches its size threshold. +- The configured flush interval elapses. +- The caller explicitly requests a WAL or memtable flush. -## Backfills plus streaming +A memtable flush first forces pending WAL data to durable storage. -If you need a full backfill of data not in the WAL along with the WAL change stream: +The database must have its WAL enabled. WAL support is enabled by default. -1. Tail the WAL, buffering changes. -2. Checkpoint the database using `Admin::create_detached_checkpoint`. -3. Range scan (`Db::scan`) over the cloned database to read old data. -4. Apply the buffered WAL changes in sequence order, filtering rows whose sequence number was in the range scan in (3). -5. Delete the checkpoint using `Admin::delete_checkpoint`. -6. Continue tailing WALs from the stored cursor. +## Basic live stream -This avoids gaps between the backfill and the live stream. Your sink should be idempotent if the backfill and WAL stream can overlap. +This example creates one unbounded iterator, emits both existing and later writes through it, and advances a durable file cursor. -## Resumability and ordering + -`WalReader` lists files in wal ID order, and each WAL file stores entries in increasing sequence order. This gives you a total order that is stable across restarts. A minimal cursor includes: +Persisting the cursor only after a complete batch gives at-least-once delivery if a process fails while emitting that WAL file. Make the sink idempotent or transactionally couple sink writes with cursor persistence when duplicates are unacceptable. -- `wal_id`: the WAL file currently being processed -- `last_seq`: the highest sequence number processed in that file +## Ordering and resumability -If you need to fan out to multiple workers, partition by key and keep one cursor per partition. Ensure your sink can handle replays or duplicates. Since WAL files can be deleted between `list` and `iterator` calls, handle `None` by skipping that file or re-listing from a newer cursor. +WAL file IDs increase monotonically, and entries within each file are returned in sequence order. The minimal cursor is therefore the last fully consumed WAL file ID. -## Listing costs and polling strategy +Do not advance the cursor per row. Doing so can lose progress for empty fence WALs and can incorrectly mark a partially emitted WAL file as complete. -The `list()` API can become expensive when WAL retention is high or GC is not keeping up. If the GC is not running, listings can grow without bound. Even with GC, CDC often needs higher retention. Retaining WAL files for just 1 hour can yield tens of thousands of files, which is expensive to list in both cost (object-store listing calls) and time. +If a requested file has already been garbage collected, iteration reports a truncation/data error. The consumer must not silently skip the gap. -If you plan to poll frequently: +## Backfills plus streaming -1. Use `list()` once to get an initial view (or to recover after a long outage). -2. Track the highest WAL ID you have successfully processed. -3. From then on, poll using `WalReader::get(latest_id + 1)`. +For a full backfill: -## Deletes, merges, and TTL +1. Begin live WAL streaming and buffer changes. +2. Create a detached database checkpoint. +3. Scan the checkpoint or a clone for existing data. +4. Apply buffered WAL changes in sequence order. +5. Delete the checkpoint. +6. Continue streaming from the persisted WAL cursor. -- Deletes appear as `ValueDeletable::Tombstone`. Emit a delete event in your CDC sink. -- Merge operands appear as `ValueDeletable::Merge`. If you use merge operators, downstream consumers must either apply the merge or store the operand as-is. -- TTL is represented by `expire_ts`. If your downstream system enforces TTL, honor this field. +The sink should tolerate overlap between the backfill and WAL stream. ## WAL retention and GC -WAL SSTs are garbage collected based on the GC configuration. If your CDC consumer falls behind, the WAL files it needs may be deleted. Set the WAL GC `min_age` option to retain files long enough for your slowest consumer and implement monitoring. Slow consumers should copy the WAL files elsewhere for processing. See [Garbage Collection](/docs/design/gc) for details. +WAL SSTs are garbage collected according to the configured retention policy. Set WAL GC minimum age long enough for the slowest consumer and monitor cursor lag. A consumer that requests a deleted WAL receives a truncation error instead of silently continuing. -## Using a separate WAL object store +## Dedicated WAL object stores -If your database uses a dedicated WAL object store, pass that store to `WalReader::new` rather than the main store. +When a database uses a dedicated WAL object store, construct the reader with both stores: the main object store is used for manifest and GC-cutoff state, and the dedicated WAL object store is used for WAL SSTs. In Rust, use **SlateDbWalReader::new_for_db_with_wal_object_store**. diff --git a/website/src/content/docs/docs/design/time.mdx b/website/src/content/docs/docs/design/time.mdx index 29941251b5..28704b4369 100644 --- a/website/src/content/docs/docs/design/time.mdx +++ b/website/src/content/docs/docs/design/time.mdx @@ -33,18 +33,18 @@ SlateDB also preserves the same metadata in the WAL path, which is why [Change D ## Expiration -TTL is stored as an absolute expiration timestamp, not a relative duration. On commit, SlateDB computes `expire_ts = create_ts + ttl`. +TTL is stored as an absolute expiration timestamp in milliseconds since the Unix epoch, not as a relative duration. On commit, SlateDB computes `expire_ts = create_ts + ttl_millis`. -[`Settings::default_ttl`](https://docs.rs/slatedb/latest/slatedb/config/struct.Settings.html#structfield.default_ttl) sets a default TTL for puts and merges. [`PutOptions`](https://docs.rs/slatedb/latest/slatedb/config/struct.PutOptions.html) and [`MergeOptions`](https://docs.rs/slatedb/latest/slatedb/config/struct.MergeOptions.html) can override that per operation: +[`Settings::default_ttl_millis`](https://docs.rs/slatedb/latest/slatedb/config/struct.Settings.html#structfield.default_ttl_millis) sets a default TTL in milliseconds for puts and merges. [`PutOptions`](https://docs.rs/slatedb/latest/slatedb/config/struct.PutOptions.html) and [`MergeOptions`](https://docs.rs/slatedb/latest/slatedb/config/struct.MergeOptions.html) can override that per operation: - [`Ttl::NoExpiry`](https://docs.rs/slatedb/latest/slatedb/config/enum.Ttl.html#variant.NoExpiry) — store the value without expiration -- [`Ttl::ExpireAfter(u64)`](https://docs.rs/slatedb/latest/slatedb/config/enum.Ttl.html#variant.ExpireAfter) — expire after a relative duration (clock ticks) -- [`Ttl::ExpireAt(i64)`](https://docs.rs/slatedb/latest/slatedb/config/enum.Ttl.html#variant.ExpireAt) — expire at a fixed absolute timestamp (clock ticks) +- [`Ttl::ExpireAfterMillis(u64)`](https://docs.rs/slatedb/latest/slatedb/config/enum.Ttl.html#variant.ExpireAfterMillis) — expire after a relative duration in milliseconds +- [`Ttl::ExpireAtMillis(i64)`](https://docs.rs/slatedb/latest/slatedb/config/enum.Ttl.html#variant.ExpireAtMillis) — expire at a fixed Unix timestamp in milliseconds Deletes write tombstones and do not carry TTL. :::caution -Batch-local merges requires one effective `expire_ts` per key. `Db::write(...)` and transaction commit return `ErrorKind::Invalid` if a single batch tries to merge the same key across different expiration timestamps. Effective timestamps are computed after applying `Settings::default_ttl`, so `Ttl::Default` can still differ from `Ttl::NoExpiry` or from another explicit TTL. +Batch-local merges requires one effective `expire_ts` per key. `Db::write(...)` and transaction commit return `ErrorKind::Invalid` if a single batch tries to merge the same key across different expiration timestamps. Effective timestamps are computed after applying `Settings::default_ttl_millis`, so `Ttl::Default` can still differ from `Ttl::NoExpiry` or from another explicit TTL. ::: :::note diff --git a/website/src/content/docs/docs/design/writes.mdx b/website/src/content/docs/docs/design/writes.mdx index af15a2318a..f99eb442e0 100644 --- a/website/src/content/docs/docs/design/writes.mdx +++ b/website/src/content/docs/docs/design/writes.mdx @@ -2,7 +2,7 @@ title: Writes --- -Writes are moved to a background task as quickly as possible to prevent blocking the client. By default, `put()`, `write()`, and `delete()` use `WriteOptions::default()`, so calls wait for the write to become durable before returning. If you want lower latency and can tolerate losing in-flight data, set `await_durable` to `false`. You can also call `flush()` explicitly. The synchronous flow is as follows: +Writes are moved to a background task as quickly as possible to prevent blocking the client. `put()`, `write()`, and `delete()` return a `WriteHandle` after the write reaches the in-memory WAL and MemTable. Call `handle.await_durable().await` to wait for that write to reach object storage, or call `flush()` explicitly. The synchronous flow is as follows: 1. A `put()`, `write()`, or `delete()` call is made on the client. 2. The key/value pair is written to the mutable, in-memory WAL table. @@ -10,8 +10,8 @@ Writes are moved to a background task as quickly as possible to prevent blocking The following asynchronous flows occur: -- The WAL flusher periodically checks if the WAL table is full. If it is, it freezes the mutable WAL table and triggers an asynchronous write to object storage. A notification is then sent to clients that wrote with `await_durable` set to `true`. -- The MemTable flusher periodically checks if the MemTable is full. If it is, it freezes the mutable MemTable and triggers an asynchronous write to object storage. A notification is then sent to clients that wrote with `await_durable` set to `true` and `wal_enabled` set to `false`. +- The WAL flusher periodically checks if the WAL table is full. If it is, it freezes the mutable WAL table and triggers an asynchronous write to object storage. Durability waiters are notified when the durable sequence number advances. +- The MemTable flusher periodically checks if the MemTable is full. If it is, it freezes the mutable MemTable and triggers an asynchronous write to object storage. With the WAL disabled, this advances the durable sequence number and notifies durability waiters. Below is a diagram illustrating the high-level flow of a write in SlateDB: @@ -72,4 +72,4 @@ To avoid this, funnel all writes that use `seqnum` through a single ordering age - **Non-contiguous seqnums.** Auto-assigned seqnums are not guaranteed to be strictly contiguous either (the memtable flusher tolerates gaps), so user-supplied jumps are consistent with existing behavior. Anything that relies on dense seqnums — don't. - **Recovery.** On restart, SlateDB recovers the max seqnum from the manifest and any unflushed WAL. If you continue assigning seqnums from your external log, make sure the next value is above whatever SlateDB recovered, or your first post-restart write will be rejected. - **Transactions.** Conflict detection still works on user-supplied seqnums — the read/write set is tracked against the snapshot's seqnum and the commit seqnum, regardless of who picked them. Just keep the monotonic-increase invariant. -- **`await_durable`.** `seqnum` is independent of the `await_durable` flag; both can be set on the same `WriteOptions`. The oracle is advanced as soon as the batch is processed by the write loop, before durability is confirmed. The seqno the oracle tracks is still bumped, so if the write fails the seqno is still consumed. +- **Durability.** The oracle is advanced as soon as the batch is processed by the write loop, before durability is confirmed. The seqno is still consumed if a later durability wait fails. Use the returned `WriteHandle::await_durable()` when confirmation is required. diff --git a/website/src/content/docs/docs/get-started/faq.mdx b/website/src/content/docs/docs/get-started/faq.mdx index a33aeff719..a2a9ff7cfe 100644 --- a/website/src/content/docs/docs/get-started/faq.mdx +++ b/website/src/content/docs/docs/get-started/faq.mdx @@ -59,7 +59,7 @@ SlateDB also offers some unique features like the ability to create snapshot clo Any in-flight data that hasn't yet been flushed to object storage will be lost. -To prevent data loss, SlateDB's `put()` API will block until the data has been flushed to object storage. Client processes can block until their data has been durably written. Blocking can be disabled with [`WriteOptions`](https://docs.rs/slatedb/latest/slatedb/config/struct.WriteOptions.html) for clients that don't need this durability guarantee. +SlateDB's write APIs return after updating the in-memory WAL and MemTable. Call `handle.await_durable().await` on the returned [`WriteHandle`](https://docs.rs/slatedb/latest/slatedb/struct.WriteHandle.html) to wait for one write to reach object storage, or call `db.flush().await` to flush all pending writes. ## Does SlateDB support column families? diff --git a/website/src/content/docs/docs/operations/benchmarks.mdx b/website/src/content/docs/docs/operations/benchmarks.mdx index 4f8eb84f77..adfb7fc72c 100644 --- a/website/src/content/docs/docs/operations/benchmarks.mdx +++ b/website/src/content/docs/docs/operations/benchmarks.mdx @@ -1,18 +1,56 @@ --- title: Benchmarks -description: See nightly performance results and learn how to run your own benchmarks +description: Review SlateDB performance results and learn how to run your own benchmarks --- -SlateDB currently has two benchmarking tools: **bencher** and **microbenchmarks**. +## Release benchmarks -- [bencher](https://github.com/slatedb/slatedb/tree/main/slatedb-bencher) is a tool to benchmark put/get operations against an object store. You can configure the tool with a variety of options. See [bencher/README.md](https://github.com/slatedb/slatedb/blob/main/slatedb-bencher/README.md) for details. [benchmark-db.sh](https://github.com/slatedb/slatedb/blob/main/slatedb-bencher/benchmark-db.sh) also serves as an example. +SlateDB publishes release benchmark results at +[benchmark.slatedb.io](https://benchmark.slatedb.io). The +[slatedb/slatedb-benchmark](https://github.com/slatedb/slatedb-benchmark) +repository contains the runner and workload definitions. It also stores the +raw results published on the site. -- Microbenchmarks run with [Criterion](https://bheisler.github.io/criterion.rs/). They are located in [benches](https://github.com/slatedb/slatedb/tree/main/slatedb/benches) and run for specific internal SlateDB functions. A comment is left on all PRs when a > 200% slowdown is detected. +The release suite starts from a shared database with 300 million records, about +120 GiB of logical data. The runner applies a fixed workload catalog to each +SlateDB revision. The catalog combines relevant workloads from +[YCSB Core](https://github.com/brianfrankcooper/YCSB/wiki/Core-Workloads) and +RocksDB's +[`db_bench`](https://github.com/facebook/rocksdb/wiki/Benchmarking-tools). +SlateDB-specific cases exercise idle behavior and transaction contention. Most +workloads run 64 closed-loop clients after a five-minute warmup and record 15 +minutes of activity. -### Nightly Benchmarks +## Other benchmarking tools -We run both **bencher** and **microbenchmarks** nightly. The [results](https://github.com/slatedb/slatedb/actions/workflows/nightly.yaml) are published in the Github action summary. +- [slatedb-bencher](https://github.com/slatedb/slatedb/tree/main/slatedb-bencher) + runs configurable database, compaction, and transaction benchmarks against + an object store. Its + [README](https://github.com/slatedb/slatedb/blob/main/slatedb-bencher/README.md) + documents the command-line options, and + [benchmark-db.sh](https://github.com/slatedb/slatedb/blob/main/slatedb-bencher/benchmark-db.sh) + provides an example workload matrix. -- **Bencher** benchmarks run on [WarpBuild](https://warpbuild.com)'s [warp-ubuntu-latest-x64-16x](https://docs.warpbuild.com/cloud-runners) runners, which use [Hetzner](https://hetzner.com/) machines in Frankfurt. We use [Tigris](https://www.tigrisdata.com/) for object storage with the `auto` region setting, which resolves to Frankfurt as well. Bandwidth between WarpBuild (Hetzner) and Tigris seems to be about 500MiB/s down and 130MiB/s up. We routinely max out the bandwidth in the nightly tests. +- SlateDB uses [Criterion](https://bheisler.github.io/criterion.rs/) for + microbenchmarks of internal functions. The benchmark sources live in + [slatedb/benches](https://github.com/slatedb/slatedb/tree/main/slatedb/benches). -- **Microbenchmarks** run on [standard Linux Github action runners](https://docs.github.com/en/actions/using-github-hosted-runners/using-github-hosted-runners/about-github-hosted-runners#standard-github-hosted-runners-for-public-repositories) with the [pprof-rs](https://github.com/tikv/pprof-rs) profiler. The resulting profiler protobuf files are published to [pprof.me](https://pprof.me) and links to each microbenchmark are provided in the Github action summary. +## Nightly microbenchmarks + +The +[nightly workflow](https://github.com/slatedb/slatedb/actions/workflows/nightly.yaml) +runs the Criterion microbenchmarks on [WarpBuild](https://warpbuild.com)'s +[warp-ubuntu-latest-arm64-8x](https://docs.warpbuild.com/cloud-runners) ARM +runners. It also records profiles with [pprof-rs](https://github.com/tikv/pprof-rs) +and uploads them to [pprof.me](https://pprof.me). The GitHub Actions job summary +links to each profile. + +## Benchmarking object stores + +SlateDB benchmarks measure the database and object store together. Use +[MinIO Warp](https://github.com/minio/warp) to measure raw S3-compatible object +store performance without SlateDB in the request path. Warp runs concurrent +GET, PUT, DELETE, and mixed-request benchmarks with configurable object sizes +and concurrency. Run it from the same region and network used by your SlateDB +clients, ideally on the same machine, so the network path and client capacity +remain comparable. diff --git a/website/src/pages/blog/[...slug].astro b/website/src/pages/blog/[...slug].astro index e846aa874e..3d6d51ff13 100644 --- a/website/src/pages/blog/[...slug].astro +++ b/website/src/pages/blog/[...slug].astro @@ -10,7 +10,7 @@ export async function getStaticPaths() { const { post } = Astro.props; const { Content, headings } = await render(post); -const { title, author, authorGithub, pubDate, ogImage } = post.data; +const { title, author, authorGithub, pubDate, ogImage, canonicalUrl } = post.data; // Default the social card to the per-post image generated at build time // (src/pages/blog/og/[...slug].png.ts); frontmatter `ogImage` can override it. @@ -25,7 +25,7 @@ const dateLabel = pubDate.toLocaleDateString('en-US', { const isoDate = pubDate.toISOString(); --- - +
diff --git a/website/src/pages/index.astro b/website/src/pages/index.astro index 02788e0862..2cd2b328ba 100644 --- a/website/src/pages/index.astro +++ b/website/src/pages/index.astro @@ -104,6 +104,7 @@ const jsonLd = { // The "used by" companies — even row, all logos rendered at the same height. const usedBy = [ { name: 'Dropbox', href: 'https://www.dropbox.com/', logo: '/img/logos/dropbox.svg', scale: 1.2 }, + { name: 'Prisma', href: 'https://www.prisma.io/', logo: '/img/logos/prisma.svg' }, { name: 'HelixDB', href: 'https://www.helix-db.com/', logo: '/img/logos/helixdb.png', scale: 0.8 }, { name: 'TensorLake', href: 'https://www.tensorlake.ai/', logo: '/img/logos/tensorlake.svg', scale: 1.2 }, { name: 'S2', href: 'https://s2.dev/', logo: '/img/logos/s2.svg' }, @@ -925,6 +926,19 @@ const osLinks = [ filter: grayscale(0%); } + @media (max-width: 640px) { + .usedby-row { + display: flex; + flex-wrap: wrap; + justify-content: center; + column-gap: 0.75rem; + } + .usedby-logo { + flex: 0 0 calc((100% - 1.5rem) / 3); + width: calc((100% - 1.5rem) / 3); + } + } + /* ---------- Editorial prose ---------- */ .prose {