diff --git a/.github/compare_benchmarks.py b/.github/compare_benchmarks.py index b8ea342be..5840e493e 100644 --- a/.github/compare_benchmarks.py +++ b/.github/compare_benchmarks.py @@ -9,6 +9,7 @@ import json import sys import argparse +import statistics from typing import Dict, List, Any @@ -47,18 +48,22 @@ def get_cold_metrics(iteration_results: List[Dict[str, Any]]) -> Dict[str, float def get_warm_metrics(iteration_results: List[Dict[str, Any]]) -> Dict[str, float]: - """Calculate average metrics from warm iterations (excluding first).""" + """Calculate median metrics from warm iterations (excluding first). + + CI runners occasionally pause a process while it is being timed. A median + keeps one such pause from turning into a reported regression. + """ warm_results = iteration_results[1:] if len(iteration_results) > 1 else iteration_results if not warm_results: return {"time_millis": 0, "cache_cpu_time": 0} - avg_time = sum(r["time_millis"] for r in warm_results) / len(warm_results) - avg_cpu_time = sum(r.get("cache_cpu_time", 0) for r in warm_results) / len(warm_results) + median_time = statistics.median(r["time_millis"] for r in warm_results) + median_cpu_time = statistics.median(r.get("cache_cpu_time", 0) for r in warm_results) # No memory column in report return { - "time_millis": avg_time, - "cache_cpu_time": avg_cpu_time, + "time_millis": median_time, + "cache_cpu_time": median_cpu_time, } @@ -79,15 +84,17 @@ def format_metric_with_baseline(current: float, baseline: float, formatter_func) return f"{formatter_func(current)} *({formatter_func(baseline)})*" -def format_change_percentage(current: float, baseline: float, highlight_mode: str = "none") -> str: +def format_change_percentage( + current: float, baseline: float, threshold: float, highlight_mode: str = "none" +) -> str: """Format percentage change and optionally highlight when slower. highlight_mode: - "none": never bold - - "slower_only": bold only if current > baseline (i.e., slower) and ≥15% + - "slower_only": bold only if current exceeds baseline by the threshold """ change_pct = calculate_change(baseline, current) - if highlight_mode == "slower_only" and change_pct > 0 and abs(change_pct) >= 15.0: + if highlight_mode == "slower_only" and change_pct >= threshold: return f"**{change_pct:+.1f}%**" return f"{change_pct:+.1f}%" @@ -195,21 +202,21 @@ def extract_mode(d: Dict[str, Any]) -> str: comp['curr_cold_time'], comp['baseline_cold_time'], format_time ) cold_change_str = format_change_percentage( - comp['curr_cold_time'], comp['baseline_cold_time'], highlight_mode="none" + comp['curr_cold_time'], comp['baseline_cold_time'], threshold, highlight_mode="none" ) warm_time_str = format_metric_with_baseline( comp['curr_warm_time'], comp['baseline_warm_time'], format_time ) warm_change_str = format_change_percentage( - comp['curr_warm_time'], comp['baseline_warm_time'], highlight_mode="slower_only" + comp['curr_warm_time'], comp['baseline_warm_time'], threshold, highlight_mode="slower_only" ) cpu_time_str = format_metric_with_baseline( comp['curr_cpu_time'], comp['baseline_cpu_time'], format_time ) cpu_change_str = format_change_percentage( - comp['curr_cpu_time'], comp['baseline_cpu_time'], highlight_mode="none" + comp['curr_cpu_time'], comp['baseline_cpu_time'], threshold, highlight_mode="none" ) lines.append( @@ -220,7 +227,7 @@ def extract_mode(d: Dict[str, Any]) -> str: ) # Summary focused on LiquidCache being slower than DataFusion (warm time) - slower_warm = [c for c in comparison if c["warm_time_change"] > 0] + slower_warm = [c for c in comparison if c["warm_time_change"] >= threshold] lines.append("") if slower_warm: lines.append(f"**⚠️ LiquidCache is slower on {len(slower_warm)} queries (warm)**") @@ -230,18 +237,22 @@ def extract_mode(d: Dict[str, Any]) -> str: slower_warm, key=lambda x: x["warm_time_change"], reverse=True ) for c in slower_warm_sorted: - curr = c["curr_warm_time"]; base = c["baseline_warm_time"] + curr = c["curr_warm_time"] + base = c["baseline_warm_time"] pct = calculate_change(base, curr) lines.append( f"- Q{c['query']}: warm {pct:+.1f}% " f"({format_time(curr)} vs {format_time(base)})" ) else: - lines.append("✅ LiquidCache is faster or equal on warm time for all queries") + lines.append(f"✅ No warm-time regression met the {threshold:.0f}% threshold") lines.append("") lines.append(f"*Compared {current_mode} vs {baseline_mode} on the same runner*") - lines.append("*Cold Time: first iteration; Warm Time: average of remaining iterations.*") + lines.append( + f"*Regressions: warm-time increases of at least {threshold:.0f}%. " + "Cold Time: first iteration; Warm Time: median of remaining iterations.*" + ) return "\n".join(lines) diff --git a/.github/test_compare_benchmarks.py b/.github/test_compare_benchmarks.py new file mode 100644 index 000000000..9fc0845ff --- /dev/null +++ b/.github/test_compare_benchmarks.py @@ -0,0 +1,38 @@ +import importlib.util +import pathlib +import unittest + + +SCRIPT = pathlib.Path(__file__).with_name("compare_benchmarks.py") +SPEC = importlib.util.spec_from_file_location("compare_benchmarks", SCRIPT) +compare = importlib.util.module_from_spec(SPEC) +SPEC.loader.exec_module(compare) + + +class CompareBenchmarksTest(unittest.TestCase): + def test_warm_metrics_use_median(self): + iterations = [ + {"time_millis": 100, "cache_cpu_time": 100}, + {"time_millis": 10, "cache_cpu_time": 1}, + {"time_millis": 11, "cache_cpu_time": 2}, + {"time_millis": 500, "cache_cpu_time": 100}, + ] + + self.assertEqual( + compare.get_warm_metrics(iterations), + {"time_millis": 11, "cache_cpu_time": 2}, + ) + + def test_highlight_respects_configured_threshold(self): + self.assertEqual( + compare.format_change_percentage(114, 100, 15, "slower_only"), + "+14.0%", + ) + self.assertEqual( + compare.format_change_percentage(114, 100, 10, "slower_only"), + "**+14.0%**", + ) + + +if __name__ == "__main__": + unittest.main() diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index cdabc9a08..3d6e9d094 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -18,7 +18,7 @@ jobs: name: Basic check runs-on: ubuntu-latest steps: - - uses: actions/checkout@v4 + - uses: actions/checkout@v7 - uses: ./.github/actions/free-disk-space - uses: dtolnay/rust-toolchain@stable with: @@ -42,6 +42,13 @@ jobs: - name: Check formatting run: cargo fmt --all -- --check + # `compare_benchmarks.py` gates the benchmark comment on every PR, so + # its tests run here rather than in the benchmark job: they need none + # of that job's artifacts, and this way a regression surfaces in + # minutes instead of after a full benchmark run. + - name: Test benchmark comparison script + run: python3 -m unittest discover -s .github -p 'test_*.py' + - name: Check documentation run: cargo doc --no-deps --document-private-items env: @@ -60,7 +67,7 @@ jobs: name: Dev Tools runs-on: ubuntu-latest steps: - - uses: actions/checkout@v4 + - uses: actions/checkout@v7 - uses: ./.github/actions/free-disk-space - uses: cachix/install-nix-action@v31 with: @@ -76,7 +83,7 @@ jobs: name: Unit Test runs-on: ubuntu-latest steps: - - uses: actions/checkout@v4 + - uses: actions/checkout@v7 - uses: ./.github/actions/free-disk-space - uses: dtolnay/rust-toolchain@stable - uses: Swatinem/rust-cache@v2 @@ -96,7 +103,7 @@ jobs: - name: Generate code coverage run: cargo llvm-cov --workspace --codecov --output-path codecov.json - name: Upload coverage to Codecov - uses: codecov/codecov-action@v5 + uses: codecov/codecov-action@v7 with: token: ${{ secrets.CODECOV_TOKEN }} files: codecov.json @@ -111,7 +118,7 @@ jobs: name: macOS runs-on: macos-latest steps: - - uses: actions/checkout@v4 + - uses: actions/checkout@v7 - uses: dtolnay/rust-toolchain@stable with: components: clippy @@ -128,7 +135,7 @@ jobs: name: Shuttle Test runs-on: ubuntu-latest steps: - - uses: actions/checkout@v4 + - uses: actions/checkout@v7 - uses: ./.github/actions/free-disk-space - uses: dtolnay/rust-toolchain@stable - uses: Swatinem/rust-cache@v2 @@ -144,7 +151,7 @@ jobs: name: Address Sanitizer runs-on: ubuntu-latest steps: - - uses: actions/checkout@v4 + - uses: actions/checkout@v7 - uses: ./.github/actions/free-disk-space # Sanitizers can only run on nightly - uses: dtolnay/rust-toolchain@nightly @@ -161,7 +168,7 @@ jobs: name: ClickBench runs-on: ubuntu-latest steps: - - uses: actions/checkout@v4 + - uses: actions/checkout@v7 - uses: ./.github/actions/free-disk-space - uses: dtolnay/rust-toolchain@stable - run: sudo apt-get update && sudo apt-get install -y wget @@ -196,7 +203,7 @@ jobs: env RUST_LOG=info cargo run --bin in_process -- --manifest benchmark/clickbench/benchmark_manifest.json --bench-mode liquid --max-memory-mb 256 cargo llvm-cov report --codecov --output-path codecov_clickbench.json - name: Upload coverage to Codecov - uses: codecov/codecov-action@v5 + uses: codecov/codecov-action@v7 with: token: ${{ secrets.CODECOV_TOKEN }} files: codecov_clickbench.json @@ -206,7 +213,7 @@ jobs: name: TPC-H runs-on: ubuntu-latest steps: - - uses: actions/checkout@v4 + - uses: actions/checkout@v7 - uses: ./.github/actions/free-disk-space - uses: dtolnay/rust-toolchain@stable - run: sudo apt-get update && sudo apt-get install -y wget @@ -236,7 +243,7 @@ jobs: env RUST_LOG=info cargo run --bin in_process -- --manifest benchmark/tpch/manifest.json --bench-mode liquid --max-memory-mb 256 cargo llvm-cov report --codecov --output-path codecov_tpch.json - name: Upload coverage to Codecov - uses: codecov/codecov-action@v5 + uses: codecov/codecov-action@v7 with: token: ${{ secrets.CODECOV_TOKEN }} files: codecov_tpch.json @@ -246,7 +253,7 @@ jobs: name: TPC-DS runs-on: ubuntu-latest steps: - - uses: actions/checkout@v4 + - uses: actions/checkout@v7 - uses: ./.github/actions/free-disk-space - uses: dtolnay/rust-toolchain@stable - run: sudo apt-get update && sudo apt-get install -y wget @@ -276,7 +283,7 @@ jobs: env RUST_LOG=info cargo run --bin in_process -- --manifest benchmark/tpcds/manifest.json --bench-mode liquid --max-memory-mb 256 cargo llvm-cov report --codecov --output-path codecov_tpcds.json - name: Upload coverage to Codecov - uses: codecov/codecov-action@v5 + uses: codecov/codecov-action@v7 with: token: ${{ secrets.CODECOV_TOKEN }} files: codecov_tpcds.json @@ -286,7 +293,7 @@ jobs: name: StackOverflow runs-on: ubuntu-latest steps: - - uses: actions/checkout@v4 + - uses: actions/checkout@v7 - uses: ./.github/actions/free-disk-space - uses: dtolnay/rust-toolchain@stable - name: Install system dependencies @@ -298,7 +305,7 @@ jobs: mkdir -p benchmark/stackoverflow/data/dba mkdir -p benchmark/stackoverflow/downloads - name: Cache StackOverflow dataset - uses: actions/cache@v4 + uses: actions/cache@v6 with: path: | benchmark/stackoverflow/data/dba @@ -331,7 +338,7 @@ jobs: env RUST_LOG=info cargo run --bin in_process -- --manifest benchmark/stackoverflow/manifest.dba.json --bench-mode liquid --max-memory-mb 10 cargo llvm-cov report --codecov --output-path codecov_stackoverflow.json - name: Upload coverage to Codecov - uses: codecov/codecov-action@v5 + uses: codecov/codecov-action@v7 with: token: ${{ secrets.CODECOV_TOKEN }} files: codecov_stackoverflow.json @@ -344,7 +351,7 @@ jobs: contents: write pull-requests: write steps: - - uses: actions/checkout@v4 + - uses: actions/checkout@v7 with: fetch-depth: 0 token: ${{ secrets.GITHUB_TOKEN }} @@ -367,10 +374,20 @@ jobs: - name: Build benchmark binary run: cargo build --release --bin in_process + # The LiquidCache run reads every query's columns before the DataFusion + # run. Without an explicit warmup, Linux's page cache makes DataFusion's + # "cold" measurements warm and creates a large, order-dependent bias. + - name: Warm filesystem cache + run: | + env RUST_LOG=warn target/release/in_process \ + --manifest benchmark/clickbench/benchmark_manifest.json \ + --iteration 1 \ + --bench-mode datafusion-default + - name: Run LiquidCache benchmark (in-process) run: | mkdir -p benchmark_results - env RUST_LOG=info cargo run --release --bin in_process -- \ + env RUST_LOG=info target/release/in_process \ --manifest benchmark/clickbench/benchmark_manifest.json \ --output benchmark_results/liquid.json \ --iteration 5 \ @@ -380,7 +397,7 @@ jobs: - name: Run DataFusion benchmark (plain parquet) run: | - env RUST_LOG=info cargo run --release --bin in_process -- \ + env RUST_LOG=info target/release/in_process \ --manifest benchmark/clickbench/benchmark_manifest.json \ --output benchmark_results/parquet.json \ --iteration 5 \ @@ -388,7 +405,7 @@ jobs: - name: Run DataFusion benchmark (default config) run: | - env RUST_LOG=info cargo run --release --bin in_process -- \ + env RUST_LOG=info target/release/in_process \ --manifest benchmark/clickbench/benchmark_manifest.json \ --output benchmark_results/df_default.json \ --iteration 5 \ @@ -417,7 +434,7 @@ jobs: - name: Comment PR with benchmark results if: steps.compare.outputs.COMPARISON_AVAILABLE == 'true' && github.event_name == 'pull_request' - uses: actions/github-script@v7 + uses: actions/github-script@v9 with: script: | const fs = require('fs'); @@ -479,7 +496,7 @@ jobs: name: Run client/server/inprocess examples runs-on: ubuntu-latest steps: - - uses: actions/checkout@v4 + - uses: actions/checkout@v7 - uses: dtolnay/rust-toolchain@stable - uses: Swatinem/rust-cache@v2 with: diff --git a/.github/workflows/fuzz.yml b/.github/workflows/fuzz.yml index bf4bc824c..9d6513793 100644 --- a/.github/workflows/fuzz.yml +++ b/.github/workflows/fuzz.yml @@ -18,7 +18,7 @@ jobs: target: [fsst_view_fuzz] steps: - name: Checkout - uses: actions/checkout@v4 + uses: actions/checkout@v7 - uses: ./.github/actions/free-disk-space @@ -95,7 +95,7 @@ jobs: } >> "$GITHUB_STEP_SUMMARY" - name: Upload coverage text summary - uses: actions/upload-artifact@v4 + uses: actions/upload-artifact@v7 if: always() with: name: fuzz-coverage-text-${{ matrix.target }} @@ -104,7 +104,7 @@ jobs: if-no-files-found: warn - name: Upload fuzz artifacts (crashes, reduces, etc.) - uses: actions/upload-artifact@v4 + uses: actions/upload-artifact@v7 if: always() with: name: fuzz-artifacts-${{ matrix.target }} diff --git a/.github/workflows/prepare-release.yml b/.github/workflows/prepare-release.yml index 822ec9dfe..af042bcfe 100644 --- a/.github/workflows/prepare-release.yml +++ b/.github/workflows/prepare-release.yml @@ -30,7 +30,7 @@ jobs: pull-requests: write steps: - name: Checkout code - uses: actions/checkout@v4 + uses: actions/checkout@v7 with: fetch-depth: 0 token: ${{ secrets.GITHUB_TOKEN }} @@ -63,7 +63,7 @@ jobs: fi - name: Create Pull Request - uses: peter-evans/create-pull-request@v7 + uses: peter-evans/create-pull-request@v8 with: token: ${{ secrets.PAT_TOKEN }} branch: release/v${{ steps.bump.outputs.new_version }} diff --git a/.github/workflows/publish.yml b/.github/workflows/publish.yml index 022a5a3ff..0022c14b1 100644 --- a/.github/workflows/publish.yml +++ b/.github/workflows/publish.yml @@ -22,7 +22,7 @@ jobs: contents: write steps: - name: Checkout code - uses: actions/checkout@v4 + uses: actions/checkout@v7 with: ref: main fetch-depth: 0 diff --git a/Cargo.lock b/Cargo.lock index 62643e0cf..2b931f618 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -33,9 +33,9 @@ dependencies = [ [[package]] name = "aho-corasick" -version = "1.1.4" +version = "1.1.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ddd31a130427c27518df266943a5308ed92d4b226cc639f5a8f1002816174301" +checksum = "c982642fa9e8606056828ee9a8505737230110bb1099153c79efe865c59d12ba" dependencies = [ "memchr", ] @@ -57,9 +57,9 @@ checksum = "cc7bb162ec39d46ab1ca8c77bf72e890535becd1751bb45f64c597edb4c8c6b3" [[package]] name = "alloc-stdlib" -version = "0.2.2" +version = "0.2.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "94fb8275041c72129eb51b7d0322c29b8387a0386127718b096429201a5d6ece" +checksum = "0e76a019e91224d279006ff972f1e984179a6e9feb050adba6ce8274aef23195" dependencies = [ "alloc-no-stdlib", ] @@ -72,9 +72,9 @@ checksum = "683d7910e743518b0e34f1186f92494becacb047c7b6bf616c96772180fef923" [[package]] name = "android_system_properties" -version = "0.1.5" +version = "0.1.6" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "819e7219dbd41043ac279b19830f2efc897156490d7fd6ea916720117ee66311" +checksum = "ae221649c9976a6f6c56ae1facf410f3ddb33cc661c4b7b61020a912d4237fbc" dependencies = [ "libc", ] @@ -131,17 +131,17 @@ dependencies = [ [[package]] name = "anyhow" -version = "1.0.102" +version = "1.0.104" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7f202df86484c868dbad7eaa557ef785d5c66295e41b460ef922eca0723b842c" +checksum = "330a5ed07fa54e4702c9d6c4174f74427fc0ef6e214bbd677ae50a5099946470" [[package]] name = "ar_archive_writer" -version = "0.5.1" +version = "0.5.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7eb93bbb63b9c227414f6eb3a0adfddca591a8ce1e9b60661bb08969b87e340b" +checksum = "73cd58deff2140a0a8eae87e417bd01db68a33e148aa93d1e8cd837e55e312b6" dependencies = [ - "object", + "object 0.39.1", ] [[package]] @@ -153,17 +153,11 @@ dependencies = [ "derive_arbitrary", ] -[[package]] -name = "arrayref" -version = "0.3.9" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "76a2e8124351fda1ef8aaaa3bbd7ebbcb486bbcd4225aca0aa0d84bb2db8fecb" - [[package]] name = "arrayvec" -version = "0.7.6" +version = "0.7.8" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7c02d123df017efcdfbd739ef81735b36c5ba83ec3c59c80a9d7ecc718f92e50" +checksum = "d3fb67a6e08acf24fdeccbac2cb6ac4305825bd1f117462e0e6f2f193345ad56" [[package]] name = "arrow" @@ -338,7 +332,7 @@ dependencies = [ "arrow-select", "chrono", "half", - "indexmap 2.14.0", + "indexmap", "itoa", "lexical-core", "memchr", @@ -431,9 +425,9 @@ checksum = "bfdc70193dadb9d7287fa4b633f15f90c876915b31f6af17da307fc59c9859a8" [[package]] name = "async-compression" -version = "0.4.42" +version = "0.4.43" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e79b3f8a79cccc2898f31920fc69f304859b3bd567490f75ebf51ae1c792a9ac" +checksum = "3976abdc8fe7d1133d43d304afd42abdf5bc3e1319d263d223bde07b5efc4be8" dependencies = [ "compression-codecs", "compression-core", @@ -460,18 +454,18 @@ checksum = "c7c24de15d275a1ecfd47a380fb4d5ec9bfe0933f309ed5e705b775596a3574d" dependencies = [ "proc-macro2", "quote", - "syn 2.0.117", + "syn 2.0.119", ] [[package]] name = "async-trait" -version = "0.1.89" +version = "0.1.92" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9035ad2d096bed7955a320ee7e2230574d28fd3c3a0f186cbea1ff3c7eed5dbb" +checksum = "82f6aeea286b8eb4dd3431a1be1b59d290ace00f5bfd8e2a159bc2a05e2c1667" dependencies = [ "proc-macro2", "quote", - "syn 2.0.117", + "syn 3.0.4", ] [[package]] @@ -507,9 +501,9 @@ checksum = "1505bd5d3d116872e7271a6d4e16d81d0c8570876c8de68093a09ac269d8aac0" [[package]] name = "autocfg" -version = "1.5.0" +version = "1.5.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c08606f8c3cbf4ce6ec8e28fb0014a2c086708fe954eaa885384a6165172e7e8" +checksum = "f2032f911046de80f0a198e0901378627c33f59ea0ac00e363d481118bd70a53" [[package]] name = "axum" @@ -599,7 +593,7 @@ checksum = "7aa268c23bfbbd2c4363b9cd302a4f504fb2a9dfe7e3451d66f35dd392e20aca" dependencies = [ "proc-macro2", "quote", - "syn 2.0.117", + "syn 2.0.119", ] [[package]] @@ -611,8 +605,8 @@ dependencies = [ "addr2line", "cfg-if", "libc", - "miniz_oxide", - "object", + "miniz_oxide 0.8.9", + "object 0.37.3", "rustc-demangle", "windows-link", ] @@ -623,12 +617,6 @@ version = "0.2.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "d27c3610c36aee21ce8ac510e6224498de4228ad772a171ed65643a24693a5a8" -[[package]] -name = "base64" -version = "0.21.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9d297deb1925b89f2ccc13d7635fa0714f12c87adce1c75356b39ca9b7178567" - [[package]] name = "base64" version = "0.22.1" @@ -649,7 +637,7 @@ checksum = "4d6867f1565b3aad85681f1015055b087fcfd840d6aeee6eee7f2da317603695" dependencies = [ "autocfg", "libm", - "num-bigint 0.4.6", + "num-bigint 0.4.8", "num-integer", "num-traits", ] @@ -662,18 +650,18 @@ checksum = "bef38d45163c2f1dde094a7dfd33ccf595c92905c8f8f4fdc18d06fb1037718a" [[package]] name = "bitflags" -version = "2.11.1" +version = "2.13.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c4512299f36f043ab09a583e57bceb5a5aab7a73db1805848e8fef3c9e8c78b3" +checksum = "b588b76d00fde79687d7646a9b5bdf3cc0f655e0bbd080335a95d7e96f3587da" dependencies = [ "serde_core", ] [[package]] name = "bitvec" -version = "1.0.1" +version = "1.1.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1bc2832c24239b0141d5674bb9174f9d68a8b5b3f2753311927c172ca46f7e9c" +checksum = "ddcec3d12c579d40898fe0a9a358a803c23e9c52ca3c425707f81c9436211837" dependencies = [ "funty", "radium", @@ -692,16 +680,15 @@ dependencies = [ [[package]] name = "blake3" -version = "1.8.5" +version = "1.8.7" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0aa83c34e62843d924f905e0f5c866eb1dd6545fc4d719e803d9ba6030371fce" +checksum = "6d9e454fc11f76977dc803893aff6304ed33d6a26efae8696573bea74baa27ae" dependencies = [ - "arrayref", "arrayvec", "cc", "cfg-if", "constant_time_eq", - "cpufeatures 0.3.0", + "cpufeatures 0.3.1", ] [[package]] @@ -715,18 +702,18 @@ dependencies = [ [[package]] name = "block-buffer" -version = "0.12.0" +version = "0.12.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cdd35008169921d80bc60d3d0ab416eecb028c4cd653352907921d95084790be" +checksum = "d2f6c7dbe95a6ed67ad9f18e57daf93a2f034c524b99fd2b76d18fdfeb6660aa" dependencies = [ "hybrid-array", ] [[package]] name = "brotli" -version = "8.0.2" +version = "8.0.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4bd8b9603c7aa97359dbd97ecf258968c95f3adddd6db2f7e7a5bef101c84560" +checksum = "5cc91aac060a7a1e25823bdccbfb6af1875b88f17c6daac97894eed8207166b3" dependencies = [ "alloc-no-stdlib", "alloc-stdlib", @@ -735,9 +722,9 @@ dependencies = [ [[package]] name = "brotli-decompressor" -version = "5.0.0" +version = "5.0.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "874bb8112abecc98cbd6d81ea4fa7e94fb9449648c93cc89aa40c81c24d7de03" +checksum = "3a32acac15fe1967bc3986b2a6347dffc965602354ea6f450ad07e8bfd253583" dependencies = [ "alloc-no-stdlib", "alloc-stdlib", @@ -745,15 +732,15 @@ dependencies = [ [[package]] name = "bumpalo" -version = "3.20.2" +version = "3.20.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5d20789868f4b01b2f2caec9f5c4e0213b41e3e5702a50157d699ae31ced2fcb" +checksum = "72f5acc6cb2ba439de613abc23857ec3d78374d8ed5ac84e9d11336e87da8649" [[package]] name = "bytemuck" -version = "1.25.0" +version = "1.25.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c8efb64bd706a16a1bdde310ae86b351e4d21550d98d056f22f8a7f7a2183fec" +checksum = "95832e849adfb21180ccb6826a99da14e5d266ae5c2e668e1602cf234f153797" [[package]] name = "byteorder" @@ -763,9 +750,9 @@ checksum = "1fd0f2584146f6f2ef48085050886acf353beff7305ebd1ae69500e27c67f64b" [[package]] name = "bytes" -version = "1.11.1" +version = "1.12.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1e748733b7cbc798e1434b6ac524f0c1ff2ab456fe201501e6497c8417a4fc33" +checksum = "fc652a48c352aef3ea3aed32080501cf3ef6ed5da78602a020c991775b0aff04" dependencies = [ "serde", ] @@ -787,9 +774,9 @@ checksum = "cd17eb909a8c6a894926bfcc3400a4bb0e732f5a57d37b1f14e8b29e329bace8" [[package]] name = "cc" -version = "1.2.61" +version = "1.4.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d16d90359e986641506914ba71350897565610e87ce0ad9e6f28569db3dd5c6d" +checksum = "0ad534f4357a5264cce5019c989cf66a4f0dc4e0d1b1d15f8aacec0ff7360273" dependencies = [ "find-msvc-tools", "jobserver", @@ -822,18 +809,18 @@ checksum = "9330f8b2ff13f34540b44e946ef35111825727b38d33286ef986142615121801" [[package]] name = "cfg_aliases" -version = "0.2.1" +version = "0.2.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "613afe47fcd5fac7ccf1db93babcb082c5994d996f20b8b159f2ad1658eb5724" +checksum = "f079e83a288787bcd14a6aea84cee5c87a67c5a3e660c30f557a3d24761b3527" [[package]] name = "chacha20" -version = "0.10.0" +version = "0.10.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6f8d983286843e49675a4b7a2d174efe136dc93a18d69130dd18198a6c167601" +checksum = "65c35e4b699c7e15ccbe7ee35c005e4fc0a278d22238a2857e6ce2dadeda1b06" dependencies = [ "cfg-if", - "cpufeatures 0.3.0", + "cpufeatures 0.3.1", "rand_core 0.10.1", ] @@ -900,9 +887,9 @@ dependencies = [ [[package]] name = "clap" -version = "4.6.1" +version = "4.6.6" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1ddb117e43bbf7dacf0a4190fef4d345b9bad68dfc649cb349e7d17d28428e51" +checksum = "473c7e07f409a8d772161724aa8db6a765a2532a70f9667eeb7b49d3d02fbdca" dependencies = [ "clap_builder", "clap_derive", @@ -910,9 +897,9 @@ dependencies = [ [[package]] name = "clap_builder" -version = "4.6.0" +version = "4.6.6" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "714a53001bf66416adb0e2ef5ac857140e7dc3a0c48fb28b2f10762fc4b5069f" +checksum = "7b48fea5a88e9ae728a2dcbedbfc0e730f7d60da42e1cb049a83c9fb8b789889" dependencies = [ "anstream", "anstyle", @@ -923,14 +910,14 @@ dependencies = [ [[package]] name = "clap_derive" -version = "4.6.1" +version = "4.6.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f2ce8604710f6733aa641a2b3731eaa1e8b3d9973d5e3565da11800813f997a9" +checksum = "d012d2b9d65aca7f18f4d9878a045bc17899bba951561ba5ec3c2ba1eed9a061" dependencies = [ "heck", "proc-macro2", "quote", - "syn 2.0.117", + "syn 3.0.4", ] [[package]] @@ -945,11 +932,20 @@ version = "1.0.5" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "1d07550c9036bf2ae0c684c4297d503f838287c83c53686d05370d0e139ae570" +[[package]] +name = "colored" +version = "3.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "faf9468729b8cbcea668e36183cb69d317348c2e08e994829fb56ebfdfbaac34" +dependencies = [ + "windows-sys 0.61.2", +] + [[package]] name = "combine" -version = "4.6.7" +version = "4.6.8" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ba5a308b75df32fe02788e748662718f03fde005016435c444eea572398219fd" +checksum = "cfc320937d09e6de266b31b9afb480f197d7a861be86be7cb2ea7e5d1bfffc5e" dependencies = [ "bytes", "memchr", @@ -1003,9 +999,9 @@ dependencies = [ [[package]] name = "console" -version = "0.16.3" +version = "0.16.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d64e8af5551369d19cf50138de61f1c42074ab970f74e99be916646777f8fc87" +checksum = "4fe5f465a4f6fee88fad41b85d990f84c835335e85b5d9e6e63e0d06d28cba7c" dependencies = [ "encode_unicode", "libc", @@ -1054,7 +1050,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "9e42cd5aabba86f128b3763da1fec1491c0f728ce99245062cd49b6f9e6d235b" dependencies = [ "const-serialize 0.7.2", - "const-serialize-macro 0.8.0-alpha.0", + "const-serialize-macro 0.8.0-alpha.1", "serde", ] @@ -1066,18 +1062,18 @@ checksum = "4f160aad86b4343e8d4e261fee9965c3005b2fd6bc117d172ab65948779e4acf" dependencies = [ "proc-macro2", "quote", - "syn 2.0.117", + "syn 2.0.119", ] [[package]] name = "const-serialize-macro" -version = "0.8.0-alpha.0" +version = "0.8.0-alpha.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "42571ed01eb46d2e1adcf99c8ca576f081e46f2623d13500eba70d1d99a4c439" +checksum = "8e4c3b1c2ce89797adff100c510b92c7cf32983fdbc632e703253f2ef516ef56" dependencies = [ "proc-macro2", "quote", - "syn 2.0.117", + "syn 2.0.119", ] [[package]] @@ -1094,9 +1090,9 @@ checksum = "b0664d2867b4a32697dfe655557f5c3b187e9b605b38612a748e5ec99811d160" [[package]] name = "const_for" -version = "0.1.5" +version = "0.1.6" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d9c50fcfdf972929aff202c16b80086aa3cfc6a3a820af714096c58c7c1d0582" +checksum = "988d3bd6bf67b6d7ae2b519c55296fa57411c27c312575e47b2975f8d1d8590b" [[package]] name = "const_format" @@ -1134,6 +1130,12 @@ dependencies = [ "charset", ] +[[package]] +name = "convert_case" +version = "0.4.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6245d59a3e82a7fc217c5828a6692dbc6dfb63a0c8c90495621f7b9d79704a0e" + [[package]] name = "convert_case" version = "0.8.0" @@ -1154,9 +1156,9 @@ dependencies = [ [[package]] name = "cookie" -version = "0.18.1" +version = "0.18.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4ddef33a339a91ea89fb53151bd0a4689cfce27055c291dfa69945475d22c747" +checksum = "1a373e3602691c3cdea496d2f0ee5935151e6168fe87739483c463db1b2f2f87" dependencies = [ "percent-encoding", "time", @@ -1207,17 +1209,11 @@ version = "0.8.7" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "773648b94d0e5d620f64f280777445740e61fe701025087ec8b57f45c791888b" -[[package]] -name = "core_detect" -version = "1.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7f8f80099a98041a3d1622845c271458a2d73e688351bf3cb999266764b81d48" - [[package]] name = "corosensei" -version = "0.3.3" +version = "0.3.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2c54787b605c7df106ceccf798df23da4f2e09918defad66705d1cedf3bb914f" +checksum = "6886a0c0f263965933c438626e7179139a62b978a33aa18281cbf0cd5a975f34" dependencies = [ "autocfg", "cfg-if", @@ -1246,45 +1242,45 @@ dependencies = [ [[package]] name = "cpufeatures" -version = "0.3.0" +version = "0.3.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8b2a41393f66f16b0823bb79094d54ac5fbd34ab292ddafb9a0456ac9f87d201" +checksum = "5ca28b0ae3115b884660db4118d803791fd6756b6e88f39c0f3f7859060d7566" dependencies = [ "libc", ] [[package]] name = "crc32fast" -version = "1.5.0" +version = "1.5.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9481c1c90cbf2ac953f07c8d4a58aa3945c425b7185c9154d67a65e4230da511" +checksum = "8498c871161e1742aaa9d52551b2d6ebdd4c3d45a3be423e3728f33b955be550" dependencies = [ "cfg-if", ] [[package]] name = "crossbeam-channel" -version = "0.5.15" +version = "0.5.16" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "82b8f8f868b36967f9606790d1903570de9ceaf870a7bf9fbbd3016d636a2cb2" +checksum = "d85363c37faeca707aef026efa9f3b34d077bce547e48f770770625c6013679e" dependencies = [ "crossbeam-utils", ] [[package]] name = "crossbeam-epoch" -version = "0.9.18" +version = "0.9.20" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5b82ac4a3c2ca9c3460964f020e1402edd5753411d7737aa39c3714ad1b5420e" +checksum = "2d6914041f254d6e9176c01941b21115dcfb7089e55135a35411081bd106ef3f" dependencies = [ "crossbeam-utils", ] [[package]] name = "crossbeam-utils" -version = "0.8.21" +version = "0.8.22" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d0a5c400df2834b80a4c3327b3aad3a4c4cd4de0629063962b03235697506a28" +checksum = "61803da095bee82a81bb1a452ecc25d3b2f1416d1897eb86430c6159ef717c17" [[package]] name = "crunchy" @@ -1352,7 +1348,7 @@ dependencies = [ "ident_case", "proc-macro2", "quote", - "syn 2.0.117", + "syn 2.0.119", ] [[package]] @@ -1363,7 +1359,7 @@ checksum = "d38308df82d1080de0afee5d069fa14b0326a88c14f15c5ccda35b4a6c414c81" dependencies = [ "darling_core", "quote", - "syn 2.0.117", + "syn 2.0.119", ] [[package]] @@ -1382,9 +1378,9 @@ dependencies = [ [[package]] name = "data-encoding" -version = "2.11.0" +version = "2.11.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a4ae5f15dda3c708c0ade84bfee31ccab44a3da4f88015ed22f63732abe300c8" +checksum = "4583a4551df46e2792f82ceeac45e850d2e2d5debba0b91f102385cda5b11f06" [[package]] name = "datafusion" @@ -1424,7 +1420,7 @@ dependencies = [ "datafusion-sql", "flate2", "futures", - "indexmap 2.14.0", + "indexmap", "itertools 0.15.0", "liblzma", "log", @@ -1501,7 +1497,7 @@ dependencies = [ "foldhash 0.2.0", "half", "hashbrown 0.17.1", - "indexmap 2.14.0", + "indexmap", "itertools 0.15.0", "libc", "log", @@ -1556,7 +1552,7 @@ dependencies = [ "log", "object_store", "parking_lot", - "rand 0.9.4", + "rand 0.9.5", "tokio", "tokio-util", "url", @@ -1694,7 +1690,7 @@ dependencies = [ "object_store", "parking_lot", "pin-project-lite", - "rand 0.9.4", + "rand 0.9.5", "tempfile", "tokio", "tokio-util", @@ -1719,7 +1715,7 @@ dependencies = [ "datafusion-physical-expr-common", "datafusion-proto-common", "datafusion-proto-models", - "indexmap 2.14.0", + "indexmap", "itertools 0.15.0", "recursive", "serde_json", @@ -1734,7 +1730,7 @@ checksum = "2604994999d5aeca1d1df645ffc98bc787447aaff05dde27aad0342b48fc1fe0" dependencies = [ "arrow", "datafusion-common", - "indexmap 2.14.0", + "indexmap", "itertools 0.15.0", ] @@ -1764,7 +1760,7 @@ dependencies = [ "md-5 0.11.0", "memchr", "num-traits", - "rand 0.9.4", + "rand 0.9.5", "regex", "sha2 0.11.0", "uuid", @@ -1894,7 +1890,7 @@ dependencies = [ "datafusion-expr", "datafusion-expr-common", "datafusion-physical-expr", - "indexmap 2.14.0", + "indexmap", "itertools 0.15.0", "log", "recursive", @@ -1917,7 +1913,7 @@ dependencies = [ "datafusion-proto-models", "half", "hashbrown 0.17.1", - "indexmap 2.14.0", + "indexmap", "itertools 0.15.0", "parking_lot", "petgraph", @@ -1952,7 +1948,7 @@ dependencies = [ "datafusion-expr-common", "datafusion-proto-models", "hashbrown 0.17.1", - "indexmap 2.14.0", + "indexmap", "itertools 0.15.0", "parking_lot", "pin-project", @@ -2005,7 +2001,7 @@ dependencies = [ "futures", "half", "hashbrown 0.17.1", - "indexmap 2.14.0", + "indexmap", "itertools 0.15.0", "log", "num-traits", @@ -2107,7 +2103,7 @@ dependencies = [ "datafusion-common", "datafusion-expr", "datafusion-functions-nested", - "indexmap 2.14.0", + "indexmap", "log", "recursive", "regex", @@ -2124,14 +2120,42 @@ dependencies = [ "uuid", ] +[[package]] +name = "defmt" +version = "1.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e2953bfe4f93bbd20cc71198842756f77d161884c99ebbabc41d80231ded88d1" +dependencies = [ + "bitflags 1.3.2", + "defmt-macros", +] + +[[package]] +name = "defmt-macros" +version = "1.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bad9c72e7ca2137e0dc3813245a0d282fd6daad32fd800af018306a9169b5fe8" +dependencies = [ + "defmt-parser", + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "defmt-parser" +version = "1.0.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "10d60334b3b2e7c9d91ef8150abfb6fa4c1c39ebbcf4a81c2e346aad939fee3e" +dependencies = [ + "thiserror 2.0.20", +] + [[package]] name = "deranged" version = "0.5.8" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "7cd812cc2bc1d69d4764bd80df88b4317eaef9e773c75226407d9bc0876b211c" -dependencies = [ - "powerfmt", -] [[package]] name = "derive_arbitrary" @@ -2141,7 +2165,7 @@ checksum = "1e567bd82dcff979e4b03460c307b3cdc9e96fde3d73bed1496d2bc75d9dd62a" dependencies = [ "proc-macro2", "quote", - "syn 2.0.117", + "syn 2.0.119", ] [[package]] @@ -2163,7 +2187,7 @@ dependencies = [ "proc-macro2", "quote", "rustc_version", - "syn 2.0.117", + "syn 2.0.119", "unicode-xid", ] @@ -2172,6 +2196,7 @@ name = "dev-tools" version = "0.1.3" dependencies = [ "dioxus", + "wasm-bindgen", ] [[package]] @@ -2191,16 +2216,16 @@ version = "0.11.3" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "f1dd6dbb5841937940781866fa1281a1ff7bd3bf827091440879f9994983d5c2" dependencies = [ - "block-buffer 0.12.0", + "block-buffer 0.12.1", "const-oid", "crypto-common 0.2.2", ] [[package]] name = "dioxus" -version = "0.7.5" +version = "0.7.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a44c550c06b6785e16258ad620d5b559f5bbcbcc50e3c18c08aa6af2604a4c32" +checksum = "9320593eda4f01858f046698b42441603d2af5fcb62a6c1f9046a0c2eebd6d9b" dependencies = [ "dioxus-asset-resolver", "dioxus-cli-config", @@ -2231,9 +2256,9 @@ dependencies = [ [[package]] name = "dioxus-asset-resolver" -version = "0.7.6" +version = "0.7.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a8b546050ecfc7fcd310be344b2f3a2a79f21554c5a9e8df28e7a07e9b36009b" +checksum = "84235ff0e272f15d4537603cdd50df6d45d50e51bb96326dbc0bdfd01d825226" dependencies = [ "dioxus-cli-config", "http", @@ -2244,7 +2269,7 @@ dependencies = [ "ndk-context", "ndk-sys", "percent-encoding", - "thiserror 2.0.18", + "thiserror 2.0.20", "tokio", "wasm-bindgen-futures", "web-sys", @@ -2252,18 +2277,18 @@ dependencies = [ [[package]] name = "dioxus-cli-config" -version = "0.7.6" +version = "0.7.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d4ad73f0ff638cd27466d389cd57f0975f909b66130dc1c25d5212d4041e5352" +checksum = "322d10f47effebd85ee0662a2cb083ff920d40dd6295a4afd8cfa4f9bc2e05c6" dependencies = [ "wasm-bindgen", ] [[package]] name = "dioxus-config-macro" -version = "0.7.6" +version = "0.7.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e004bc8b958031117d373db2b5e0ab9d7e763751de129dd36f00ef7e318333cd" +checksum = "9f3a906c8219e94fa48189a793c760a4e3193cbb7791f799e9d37633fe41161d" dependencies = [ "proc-macro2", "quote", @@ -2271,15 +2296,15 @@ dependencies = [ [[package]] name = "dioxus-config-macros" -version = "0.7.6" +version = "0.7.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "808a9994a9a2623e6b6890b6cc68def24bd669177ec4713684447fb46418c256" +checksum = "20de015bc89c20f4ffc7b4f849943420308365fa66f991250b5c075261990a21" [[package]] name = "dioxus-core" -version = "0.7.6" +version = "0.7.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "247ed8d679a13232641f1c84ba22246623fae01320b4c22db225c0b4f2fa7398" +checksum = "1b232228ada232c0adc151e41ebd9ef70cc04683db097fa0dc6015979fa01722" dependencies = [ "anyhow", "const_format", @@ -2288,7 +2313,7 @@ dependencies = [ "futures-util", "generational-box", "longest-increasing-subsequence", - "rustc-hash 2.1.2", + "rustc-hash 2.1.3", "rustversion", "serde", "slab", @@ -2299,28 +2324,28 @@ dependencies = [ [[package]] name = "dioxus-core-macro" -version = "0.7.6" +version = "0.7.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "75fbe64029b90144041f8521300c7b3508e6c48caea3244de2ff5d1ade15390c" +checksum = "6256a68462222f0600f5fd7d9542cbd9f299bc19355d900f7827dd3fc9e09fdb" dependencies = [ "convert_case 0.8.0", "dioxus-rsx", "proc-macro2", "quote", - "syn 2.0.117", + "syn 2.0.119", ] [[package]] name = "dioxus-core-types" -version = "0.7.6" +version = "0.7.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cbfea5c8946e0745b254b5c33c515d81b3ba638f33c2a532ed06730392394d4d" +checksum = "96007ed6cbad951dfed82d83c581aabd95bc591f351f5666d454ddbe7845b324" [[package]] name = "dioxus-devtools" -version = "0.7.6" +version = "0.7.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "08d30370fa78266aed3f3d9119dea2de9e92a0348941dbfa777f7a669f2ea375" +checksum = "434cebb282b3f820a341582ebfd37dab907d5a25c3eeafd87802468afcd2a959" dependencies = [ "dioxus-cli-config", "dioxus-core", @@ -2331,16 +2356,16 @@ dependencies = [ "serde", "serde_json", "subsecond", - "thiserror 2.0.18", + "thiserror 2.0.20", "tracing", "tungstenite 0.28.0", ] [[package]] name = "dioxus-devtools-types" -version = "0.7.6" +version = "0.7.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3907a2b61cf56039f047da6a37317a03d9e15411753bc40e5e34a27e405ac320" +checksum = "dfd6f3475e38c93be245a664a9ce11679a8f2b99832620705cf7837f42aa39ce" dependencies = [ "dioxus-core", "serde", @@ -2349,9 +2374,9 @@ dependencies = [ [[package]] name = "dioxus-document" -version = "0.7.6" +version = "0.7.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e3b75a1809af7c13546ae4487c8b02ab80cc4d059e7db5a5d090374ceaa71d5b" +checksum = "2f6fadd5c6d886b48b762820abae1a08142a2159f8a1136c0466b4b4ed9f547e" dependencies = [ "dioxus-core", "dioxus-core-macro", @@ -2368,9 +2393,9 @@ dependencies = [ [[package]] name = "dioxus-fullstack" -version = "0.7.6" +version = "0.7.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c8ee56dd65fbf1222fa6a2749c3f821df28facf1832854b43d480b547a096f6d" +checksum = "a91d5acabd1470145bcf164cab9fd2edcf2e4fba0fa6fb0fd6dca23079fcc169" dependencies = [ "anyhow", "async-stream", @@ -2413,13 +2438,13 @@ dependencies = [ "serde_json", "serde_qs", "serde_urlencoded", - "thiserror 2.0.18", + "thiserror 2.0.20", "tokio", "tokio-stream", "tokio-tungstenite 0.28.0", "tokio-util", "tower", - "tower-http", + "tower-http 0.6.11", "tower-layer", "tracing", "tungstenite 0.27.0", @@ -2433,9 +2458,9 @@ dependencies = [ [[package]] name = "dioxus-fullstack-core" -version = "0.7.6" +version = "0.7.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d40d33a447cb158acdb61787b2ff52dd8a0f9a9f20e95e5c5fe9873f01c2b55b" +checksum = "0011e9bca8da2ae6bf02ba9ef93ba6c5fa539e402de3aeb316ad86ca77c59079" dependencies = [ "anyhow", "axum-core", @@ -2454,30 +2479,30 @@ dependencies = [ "parking_lot", "serde", "serde_json", - "thiserror 2.0.18", + "thiserror 2.0.20", "tokio", "tracing", ] [[package]] name = "dioxus-fullstack-macro" -version = "0.7.6" +version = "0.7.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7c06eb66bce50d5f47b793e6af5fc2e0a511bf2b4fa2423cf86e35023d8f17e6" +checksum = "681f39665de36af7bb724b1cc0ca9c2b7c36a2b93febf1b2911716479eac3db0" dependencies = [ "const_format", "convert_case 0.8.0", "proc-macro2", "quote", - "syn 2.0.117", + "syn 2.0.119", "xxhash-rust", ] [[package]] name = "dioxus-history" -version = "0.7.6" +version = "0.7.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c1d8024afd482956eadae2c43d0b1e73e584adb724ac09be87f268d52002387b" +checksum = "2404b77d4441e694a5e93afb5c9a729efb26a3f6920f769cfe6cfedf5c7ac210" dependencies = [ "dioxus-core", "tracing", @@ -2485,9 +2510,9 @@ dependencies = [ [[package]] name = "dioxus-hooks" -version = "0.7.6" +version = "0.7.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "233b5e168a7c38c4bf96d0f390221a72a5a28bb31ce2dcf6d2af879dc561ef42" +checksum = "9d3807a1bc039299cd37ad85f1c56ae7088210e8a38e3e877f01034d68c15fbb" dependencies = [ "dioxus-core", "dioxus-signals", @@ -2501,9 +2526,9 @@ dependencies = [ [[package]] name = "dioxus-html" -version = "0.7.6" +version = "0.7.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4abf4ad27eee650d1ab8ebe13591e8b0ee595fa5a5dd236be13a5b7b3fab678d" +checksum = "40b9a85679c4dfd8699407a9ad5041a6a4fe6be0ee386353951cd03b57a12030" dependencies = [ "async-trait", "bytes", @@ -2528,28 +2553,28 @@ dependencies = [ [[package]] name = "dioxus-html-internal-macro" -version = "0.7.6" +version = "0.7.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "025e107e677f790f4ed648a189e36f51ae014e5341902b97313e4eae626cffa2" +checksum = "500282dad5b62e91f93e127cbf41c587bf1c457d52514a75f3ff2a55fd0ea9f6" dependencies = [ "convert_case 0.8.0", "proc-macro2", "quote", - "syn 2.0.117", + "syn 2.0.119", ] [[package]] name = "dioxus-interpreter-js" -version = "0.7.6" +version = "0.7.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "57caa76427d8ec4105ccca44ff8c511055688af732a094b9fe9ef3547d2a11b2" +checksum = "faa22cc3431e7efdbae28069542a1763c977d5edbe4d4fbfb0b7667b9c9966e9" dependencies = [ "dioxus-core", "dioxus-core-types", "dioxus-html", "js-sys", "lazy-js-bundle", - "rustc-hash 2.1.2", + "rustc-hash 2.1.3", "sledgehammer_bindgen", "sledgehammer_utils", "wasm-bindgen", @@ -2559,9 +2584,9 @@ dependencies = [ [[package]] name = "dioxus-liveview" -version = "0.7.6" +version = "0.7.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "64e64d86ad604897c796fdcc1f1f42cab838258647dd47664ab637a8eb97f08e" +checksum = "8b9cfecfc0037cb85b649739acc49a09dd303d1fa1daa7af609f5796b491b71c" dependencies = [ "axum", "dioxus-cli-config", @@ -2574,11 +2599,11 @@ dependencies = [ "futures-channel", "futures-util", "generational-box", - "rustc-hash 2.1.2", + "rustc-hash 2.1.3", "serde", "serde_json", "slab", - "thiserror 2.0.18", + "thiserror 2.0.20", "tokio", "tokio-stream", "tokio-util", @@ -2587,9 +2612,9 @@ dependencies = [ [[package]] name = "dioxus-logger" -version = "0.7.6" +version = "0.7.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0cbbee192b1b12fccb444a5b04d809710cfce4d27b792129fea6c845fae7f329" +checksum = "0f36b2dac4df8c747431f77f2a6644dd033ea98e7f8ac80271ff47fd4ba8a548" dependencies = [ "dioxus-cli-config", "tracing", @@ -2599,9 +2624,9 @@ dependencies = [ [[package]] name = "dioxus-router" -version = "0.7.6" +version = "0.7.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ecdb19d7a1489ba252be9b3a07f9db92814020eb4ba8c1306f04242a44d17e66" +checksum = "e9652a5bcead34f687d7dbb12d5c8f6a7ecb342786273395ca0078e51b24320f" dependencies = [ "dioxus-cli-config", "dioxus-core", @@ -2620,9 +2645,9 @@ dependencies = [ [[package]] name = "dioxus-router-macro" -version = "0.7.6" +version = "0.7.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b525ab775585f1dc4850178de9d0cb6bb37f7b9c1a1a0c121eab7449d7021480" +checksum = "56d20cf2173183946fdf8ab069b606a840a9d06e949050619355fc8cb97aafb6" dependencies = [ "base16", "digest 0.10.7", @@ -2630,27 +2655,27 @@ dependencies = [ "quote", "sha2 0.10.9", "slab", - "syn 2.0.117", + "syn 2.0.119", ] [[package]] name = "dioxus-rsx" -version = "0.7.6" +version = "0.7.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "37fb07e40e9734946511659668ea3675ed214b60889e7aa7f0a5a85271518475" +checksum = "aea662c9e13a91279d8acbc0d02a80430c7081704c20e8c791e4c4a151268596" dependencies = [ "proc-macro2", "proc-macro2-diagnostics", "quote", "rustversion", - "syn 2.0.117", + "syn 2.0.119", ] [[package]] name = "dioxus-server" -version = "0.7.6" +version = "0.7.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5aa22ad381073c68a70cdb28152485abb5f2dcf20dabc68c631e26ce7ac046dd" +checksum = "ad8d8cfcc78453d8f1eba4350126161844a0b10b4dc898bac7649b8615f92c48" dependencies = [ "anyhow", "async-trait", @@ -2687,17 +2712,17 @@ dependencies = [ "lru", "parking_lot", "pin-project", - "rustc-hash 2.1.2", + "rustc-hash 2.1.3", "serde", "serde_json", "serde_qs", "subsecond", - "thiserror 2.0.18", + "thiserror 2.0.20", "tokio", "tokio-tungstenite 0.28.0", "tokio-util", "tower", - "tower-http", + "tower-http 0.6.11", "tracing", "tracing-futures", "url", @@ -2706,37 +2731,37 @@ dependencies = [ [[package]] name = "dioxus-signals" -version = "0.7.6" +version = "0.7.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5393fd579f42c6547bf47ec0a2dedf2a366cd541d31deedc1059a096e6c35798" +checksum = "95d85e36fcaf7abdc986836801bf248776da7d5961f5fac65fdbf5ed491afb4f" dependencies = [ "dioxus-core", "futures-channel", "futures-util", "generational-box", "parking_lot", - "rustc-hash 2.1.2", + "rustc-hash 2.1.3", "tracing", "warnings", ] [[package]] name = "dioxus-ssr" -version = "0.7.6" +version = "0.7.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d17c75a43e218012b63a97f55cf585747bd0ca37840ac2a0cf40add06ffcf9fe" +checksum = "4544ecf908962909de403bfdf94f771b036e9e1857e371cd3279841a15a892e6" dependencies = [ "askama_escape", "dioxus-core", "dioxus-core-types", - "rustc-hash 2.1.2", + "rustc-hash 2.1.3", ] [[package]] name = "dioxus-stores" -version = "0.7.6" +version = "0.7.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f1c59f52e8194439604dd35f8c540456c22b4a4b076930e424d0289f98ea3cb4" +checksum = "15eaa9f2ccc5ce962d056515e61345c846a54b13a2b880cf7e8ac93faa53a178" dependencies = [ "dioxus-core", "dioxus-signals", @@ -2746,21 +2771,21 @@ dependencies = [ [[package]] name = "dioxus-stores-macro" -version = "0.7.6" +version = "0.7.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "737600865572cecf60ff934f88252bd4144f4ca189e0e570b7a780e8f0e01a1f" +checksum = "a13bd8a693d7042544fa3ac79698a08783bbc698d1ed257e88c5d4059e9ac476" dependencies = [ "convert_case 0.8.0", "proc-macro2", "quote", - "syn 2.0.117", + "syn 2.0.119", ] [[package]] name = "dioxus-web" -version = "0.7.6" +version = "0.7.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "176eb0a5ee8251203a816b413a64fdc014b43e53d4245f769b1b0c2035b88ac3" +checksum = "481093233d789c5ffbdde66e7970e803b01a199a7ab0c0f8ef84c65d30b7384f" dependencies = [ "dioxus-cli-config", "dioxus-core", @@ -2778,7 +2803,7 @@ dependencies = [ "gloo-timers", "js-sys", "lazy-js-bundle", - "rustc-hash 2.1.2", + "rustc-hash 2.1.3", "send_wrapper", "serde", "serde-wasm-bindgen", @@ -2790,15 +2815,25 @@ dependencies = [ "web-sys", ] +[[package]] +name = "dispatch2" +version = "0.3.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1e0e367e4e7da84520dedcac1901e4da967309406d1e51017ae1abfb97adbd38" +dependencies = [ + "bitflags 2.13.1", + "objc2", +] + [[package]] name = "displaydoc" -version = "0.2.5" +version = "0.2.7" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "97369cbbc041bc366949bc74d34658d6cda5621039731c6310521892a3a20ae0" +checksum = "c6232dd377dcc64799954cbd3a9bb882e9cdc1308ccd87b1c098f1fb2eaf82a8" dependencies = [ "proc-macro2", "quote", - "syn 2.0.117", + "syn 3.0.4", ] [[package]] @@ -2823,7 +2858,7 @@ checksum = "9556bc800956545d6420a640173e5ba7dfa82f38d3ea5a167eb555bc69ac3323" dependencies = [ "proc-macro2", "quote", - "syn 2.0.117", + "syn 2.0.119", ] [[package]] @@ -2845,7 +2880,7 @@ dependencies = [ "pretty-hex", "serde", "serde_json", - "thiserror 2.0.18", + "thiserror 2.0.20", "zerocopy", ] @@ -2857,7 +2892,7 @@ checksum = "dc09b90bda5770641457f1c0a42c8203c48f5a3d9799dcf1bafbd84e30ccf080" dependencies = [ "pest", "pest_derive", - "thiserror 2.0.18", + "thiserror 2.0.20", ] [[package]] @@ -2868,9 +2903,9 @@ checksum = "92773504d58c093f6de2459af4af33faa518c13451eb8f2b5698ed3d36e7c813" [[package]] name = "either" -version = "1.15.0" +version = "1.18.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "48c757948c5ede0e46177b7add2e67155f70e33c07fea8284df6576da70b3719" +checksum = "252afb9ae5eaa683babdc6a068b3f5726eb19e05070c731f9b2a23a7c3e8ed34" [[package]] name = "encode_unicode" @@ -2889,23 +2924,23 @@ dependencies = [ [[package]] name = "enumset" -version = "1.1.10" +version = "1.1.14" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "25b07a8dfbbbfc0064c0a6bdf9edcf966de6b1c33ce344bdeca3b41615452634" +checksum = "ccc5801fd11762e24d1e420d01d2ac518f2a2ca4329d4fbb6639f2412b6204e0" dependencies = [ "enumset_derive", ] [[package]] name = "enumset_derive" -version = "0.14.0" +version = "0.15.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f43e744e4ea338060faee68ed933e46e722fb7f3617e722a5772d7e856d8b3ce" +checksum = "4bd536557b58c682b217b8fb199afdff47cd3eff260623f19e77074eb073d63a" dependencies = [ "darling", "proc-macro2", "quote", - "syn 2.0.117", + "syn 2.0.119", ] [[package]] @@ -2925,7 +2960,7 @@ checksum = "44f23cf4b44bfce11a86ace86f8a73ffdec849c9fd00a386a53d278bd9e81fb3" dependencies = [ "proc-macro2", "quote", - "syn 2.0.117", + "syn 2.0.119", ] [[package]] @@ -2934,17 +2969,6 @@ version = "1.0.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "877a4ace8713b0bcf2a4e7eec82529c029f1d0619886d18145fea96c3ffe5c0f" -[[package]] -name = "erased-serde" -version = "0.4.10" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d2add8a07dd6a8d93ff627029c51de145e12686fbc36ecb298ac22e74cf02dec" -dependencies = [ - "serde", - "serde_core", - "typeid", -] - [[package]] name = "errno" version = "0.3.14" @@ -2996,50 +3020,47 @@ dependencies = [ [[package]] name = "fastlanes" -version = "0.5.1" +version = "0.7.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "20c597e23b8ec8506f589d18bc701ca83a3def6086748f628ad23092e1dfe577" +checksum = "34f6c951d711d8a10f08524071f6171bc95c5f857e8b39d91e36ad6353fc4f4f" dependencies = [ - "arrayref", "const_for", - "core_detect", "num-traits", - "paste", + "pastey", "seq-macro", ] [[package]] name = "fastrace" -version = "0.7.17" +version = "0.7.19" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2130caec636d7a1d23b173576674ced1af967228642ceaeb6a1b4705c282b00e" +checksum = "dcef89a7c5b37f7c13551af01f509cbd5d32a8644c27423623639eb923cc2973" dependencies = [ "fastant", "fastrace-macro", "parking_lot", "pin-project", - "rand 0.9.4", + "rand 0.10.2", "rtrb", "serde", ] [[package]] name = "fastrace-macro" -version = "0.7.17" +version = "0.7.19" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0b35f67e02527fca6515ff61f922360df781f477daf6a806fff16bd59525dee5" +checksum = "2f1fdb97ada2fde7912c9cab29b560d550ccc467ca1f39c952ce4b05f8d152b3" dependencies = [ - "proc-macro-error2", "proc-macro2", "quote", - "syn 2.0.117", + "syn 3.0.4", ] [[package]] name = "fastrace-opentelemetry" -version = "0.16.0" +version = "0.18.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4644a8a7ce6d20d83e73d9562f323388e4d817470d40bcb51fa2fe328db58cd2" +checksum = "67d87b05a82ea59d6d2a11066f4e0cc1e169ce50bc8c276279249ba374936d91" dependencies = [ "fastrace", "log", @@ -3062,15 +3083,15 @@ dependencies = [ [[package]] name = "fastrand" -version = "2.4.1" +version = "2.5.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9f1f227452a390804cdb637b74a86990f2a7d7ba4b7d5693aac9b4dd6defd8d6" +checksum = "da7c62ceae207dd37ea5b845da6a0696c799f85e97da1ab5b7910be3c1c80223" [[package]] name = "find-msvc-tools" -version = "0.1.9" +version = "0.1.11" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5baebc0774151f905a1a2cc41989300b1e6fbb29aff0ceffa1064fdd3088d582" +checksum = "d45db016d36b838f563236e9193d0ee6ce38f3f68b6c94e914b4929c96bbb890" [[package]] name = "findshlibs" @@ -3096,18 +3117,18 @@ version = "25.12.19" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "35f6839d7b3b98adde531effaf34f0c2badc6f4735d26fe74709d8e513a96ef3" dependencies = [ - "bitflags 2.11.1", + "bitflags 2.13.1", "rustc_version", ] [[package]] name = "flate2" -version = "1.1.9" +version = "1.1.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "843fba2746e448b37e26a819579957415c8cef339bf08564fe8b7ddbd959573c" +checksum = "6e634e2e0ebac1ee034020da1ca582e17ffe4e0f5e985823721e168928136dcb" dependencies = [ "crc32fast", - "miniz_oxide", + "miniz_oxide 0.9.1", "zlib-rs", ] @@ -3140,11 +3161,11 @@ dependencies = [ [[package]] name = "fsst-rs" -version = "0.5.11" +version = "0.6.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9b13ac798afc0d9194eb4efefef8b9332efbd80b43f302a968cb8cb23b9d5360" +checksum = "03b614a2a7f2efc50d76c1e47d64b57c55acf6e0e718b0cae78236d147c487d0" dependencies = [ - "rustc-hash 2.1.2", + "rustc-hash 2.1.3", ] [[package]] @@ -3155,9 +3176,9 @@ checksum = "e6d5a32815ae3f33302d95fdcb2ce17862f8c65363dcfd29360480ba1001fc9c" [[package]] name = "futures" -version = "0.3.32" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8b147ee9d1f6d097cef9ce628cd2ee62288d963e16fb287bd9286455b241382d" +checksum = "9a31d2a3fbaaeb2af2368bbdd904aa8e812d3c04a1ee10d3171f52d556e5d0a3" dependencies = [ "futures-channel", "futures-core", @@ -3170,9 +3191,9 @@ dependencies = [ [[package]] name = "futures-channel" -version = "0.3.32" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "07bbe89c50d7a535e539b8c17bc0b49bdb77747034daa8087407d655f3f7cc1d" +checksum = "b1f9e3d69d39e4862ffed03ed071a76f9a13ba1d9109d355b0f0aa6b15e393c4" dependencies = [ "futures-core", "futures-sink", @@ -3180,15 +3201,15 @@ dependencies = [ [[package]] name = "futures-core" -version = "0.3.32" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7e3450815272ef58cec6d564423f6e755e25379b217b0bc688e295ba24df6b1d" +checksum = "92d699e522242e69e3003b94ecc1f960f3a5e015aa7c5d7486e65ad01dd94f5e" [[package]] name = "futures-executor" -version = "0.3.32" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "baf29c38818342a3b26b5b923639e7b1f4a61fc5e76102d4b1981c6dc7a7579d" +checksum = "031b47cf1a3c6cc8bc2fc76cd437f521619387907d469316e7c0bc278f1f5432" dependencies = [ "futures-core", "futures-task", @@ -3197,38 +3218,38 @@ dependencies = [ [[package]] name = "futures-io" -version = "0.3.32" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cecba35d7ad927e23624b22ad55235f2239cfa44fd10428eecbeba6d6a717718" +checksum = "53c0fa8157de1303bfffdaa1cc2a673bfffb60102f76b0ef4441659124373fed" [[package]] name = "futures-macro" -version = "0.3.32" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e835b70203e41293343137df5c0664546da5745f82ec9b84d40be8336958447b" +checksum = "9fb9654ba8355388abeb8dcb4fc62f511300867002afc858860463bdd9fe0c44" dependencies = [ "proc-macro2", "quote", - "syn 2.0.117", + "syn 3.0.4", ] [[package]] name = "futures-sink" -version = "0.3.32" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c39754e157331b013978ec91992bde1ac089843443c49cbc7f46150b0fad0893" +checksum = "1944426bf7d03f1d14f708785e4b33efd750b36d48a157b836b3efc15ede8e1d" [[package]] name = "futures-task" -version = "0.3.32" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "037711b3d59c33004d3856fbdc83b99d4ff37a24768fa1be9ce3538a1cde4393" +checksum = "cd417de3d1d015fc3bfd2b1ea46dfc7bab72ef86f1cc7cc9c78e728b34a6d1fd" [[package]] name = "futures-util" -version = "0.3.32" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "389ca41296e6190b48053de0321d02a77f32f8a5d2461dd38762c0593805c6d6" +checksum = "0d50a92467f8ba5dd6e3ee5d4bd04d73ab2e4e1c44474a0674821dfce14b79bc" dependencies = [ "futures-channel", "futures-core", @@ -3243,9 +3264,9 @@ dependencies = [ [[package]] name = "generational-box" -version = "0.7.6" +version = "0.7.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c68d74be1fbe3bba37604bdfd61403f26af9f6324cf325053abd89d60c22e799" +checksum = "b8b2dc9b873cd1a8adb80aa5dcef9a8205f781a8a313d5ffb867248cb3bcb764" dependencies = [ "parking_lot", "tracing", @@ -3281,25 +3302,23 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "899def5c37c4fd7b2664648c28120ecec138e4d395b459e5ca34f9cce2dd77fd" dependencies = [ "cfg-if", - "js-sys", "libc", "r-efi 5.3.0", "wasip2", - "wasm-bindgen", ] [[package]] name = "getrandom" -version = "0.4.2" +version = "0.4.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0de51e6874e94e7bf76d726fc5d13ba782deca734ff60d5bb2fb2607c7406555" +checksum = "300e883d756b2e4ec94e02791f39b04b522276138852cfc41d9fb7e904106099" dependencies = [ "cfg-if", + "js-sys", "libc", "r-efi 6.0.0", "rand_core 0.10.1", - "wasip2", - "wasip3", + "wasm-bindgen", ] [[package]] @@ -3310,9 +3329,9 @@ checksum = "e629b9b98ef3dd8afe6ca2bd0f89306cec16d43d907889945bc5d6687f2f13c7" [[package]] name = "glob" -version = "0.3.3" +version = "0.3.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0cc23270f6e1808e30a928bdc84dea0b9b4136a8bc82338574f23baf47bbd280" +checksum = "e4eba85ea1d0a966a983acd07deee566e67395d2d96b6fb39e62b5a833f1eb0b" [[package]] name = "gloo-net" @@ -3362,9 +3381,9 @@ dependencies = [ [[package]] name = "goblin" -version = "0.10.5" +version = "0.10.7" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "983a6aafb3b12d4c41ea78d39e189af4298ce747353945ff5105b54a056e5cd9" +checksum = "17582616a7718cca54cec18e534a76c7c4aec11a8b9a85695712f262fd15a4c8" dependencies = [ "log", "plain", @@ -3373,9 +3392,9 @@ dependencies = [ [[package]] name = "h2" -version = "0.4.13" +version = "0.4.19" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2f44da3a8150a6703ed5d34e164b875fd14c2cdab9af1252a9a1020bde2bdc54" +checksum = "ef8e5e5a340588f4452631496976cf8636d4a7ecf600239fdc27615d2530bc16" dependencies = [ "atomic-waker", "bytes", @@ -3383,7 +3402,7 @@ dependencies = [ "futures-core", "futures-sink", "http", - "indexmap 2.14.0", + "indexmap", "slab", "tokio", "tokio-util", @@ -3402,12 +3421,6 @@ dependencies = [ "zerocopy", ] -[[package]] -name = "hashbrown" -version = "0.12.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8a9ee70c43aaf417c914396645a0fa852624801b24ebb7ae78fe8272889ac888" - [[package]] name = "hashbrown" version = "0.14.5" @@ -3447,11 +3460,11 @@ dependencies = [ [[package]] name = "hdrhistogram" -version = "7.5.4" +version = "7.6.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "765c9198f173dd59ce26ff9f95ef0aafd0a0fe01fb9d72841bc5066a4c06511d" +checksum = "f49d1053f4708f0af3cf9fc5bffc7e68a914a3c45becb231c80068c9c3f78bea" dependencies = [ - "base64 0.21.7", + "base64 0.22.1", "byteorder", "crossbeam-channel", "flate2", @@ -3491,9 +3504,9 @@ checksum = "2304e00983f87ffb38b55b444b5e3b60a884b5d30c0fca7d82fe33449bbe55ea" [[package]] name = "hermit-abi" -version = "0.5.2" +version = "0.5.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fc0fef456e4baa96da950455cd02c081ca953b141298e41db3fc7e36b1da849c" +checksum = "e17592d60ebacc7d5e169f4663c5f84f9161cc90328abcfe8456f41e4dfcb284" [[package]] name = "hex" @@ -3503,9 +3516,9 @@ checksum = "7f24254aa9a54b5c858eaee2f5bccdb46aaf0e486a595ed5fd8f86ba55232a70" [[package]] name = "http" -version = "1.4.0" +version = "1.5.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e3ba2a386d7f85a81f119ad7498ebe444d2e22c2af0b86b069416ace48b3311a" +checksum = "918d3568bebf352712bc2ef3d46a8bcf1a75b373be6539de198e9105cbbf9ce0" dependencies = [ "bytes", "itoa", @@ -3513,9 +3526,9 @@ dependencies = [ [[package]] name = "http-body" -version = "1.0.1" +version = "1.1.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1efedce1fb8e6913f23e0c92de8e62cd5b772a67e7b3946df930a62566c93184" +checksum = "ca2a8f2913ee65f60facd6a5905613afaa448497a0230cc41ce022d93290bc2c" dependencies = [ "bytes", "http", @@ -3523,9 +3536,9 @@ dependencies = [ [[package]] name = "http-body-util" -version = "0.1.3" +version = "0.1.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b021d93e26becf5dc7e1b75b1bed1fd93124b374ceb73f43d4d4eafec896a64a" +checksum = "23169fe34a5fbcdd3f3862e78fb9b6fccd5f02a6dc6f732547005d45631ce71c" dependencies = [ "bytes", "futures-core", @@ -3554,24 +3567,24 @@ checksum = "df3b46402a9d5adb4c86a0cf463f42e19994e3ee891101b1841f30a545cb49a9" [[package]] name = "humantime" -version = "2.3.0" +version = "2.4.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "135b12329e5e3ce057a9f972339ea52bc954fe1e9358ef27f95e89716fbc5424" +checksum = "15cdd26707701c53297e2fa6afb323d55fbc1d0810c3aec078ae3ef0424c3c15" [[package]] name = "hybrid-array" -version = "0.4.12" +version = "0.4.14" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9155a582abd142abc056962c29e3ce5ff2ad5469f4246b537ed42c5deba857da" +checksum = "707114b52a152fa7bdb290cd7cd5912d9467273b6d74e21b8d81aca1f8533f6b" dependencies = [ "typenum", ] [[package]] name = "hyper" -version = "1.9.0" +version = "1.11.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6299f016b246a94207e63da54dbe807655bf9e00044f73ded42c3ac5305fbcca" +checksum = "27b501faa50e7a26c3d3560ca625132f4078a17771f4810baf70475ae48cbe43" dependencies = [ "atomic-waker", "bytes", @@ -3671,9 +3684,9 @@ dependencies = [ [[package]] name = "icu_collections" -version = "2.2.0" +version = "2.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2984d1cd16c883d7935b9e07e44071dca8d917fd52ecc02c04d5fa0b5a3f191c" +checksum = "fa68d21081c4a05d5a901a1c62add574c77048b6a1c67be3b50ce0b60d4ca513" dependencies = [ "displaydoc", "potential_utf", @@ -3685,9 +3698,9 @@ dependencies = [ [[package]] name = "icu_locale_core" -version = "2.2.0" +version = "2.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "92219b62b3e2b4d88ac5119f8904c10f8f61bf7e95b640d25ba3075e6cac2c29" +checksum = "d56e28588da92eee5c3201a6eff33fabdd49b62269c8938d4ff050ce4d900deb" dependencies = [ "displaydoc", "litemap", @@ -3698,9 +3711,9 @@ dependencies = [ [[package]] name = "icu_normalizer" -version = "2.2.0" +version = "2.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c56e5ee99d6e3d33bd91c5d85458b6005a22140021cc324cea84dd0e72cff3b4" +checksum = "12f9cf5f235641ed274641dd81c3f28d870e276763d0797aeeab72317b1c646f" dependencies = [ "icu_collections", "icu_normalizer_data", @@ -3712,16 +3725,17 @@ dependencies = [ [[package]] name = "icu_normalizer_data" -version = "2.2.0" +version = "2.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "da3be0ae77ea334f4da67c12f149704f19f81d1adf7c51cf482943e84a2bad38" +checksum = "1563da1ed3e0b3bf3d74c9b85917ac9c56464d2f57242270c09c9e752f8021a0" [[package]] name = "icu_properties" -version = "2.2.0" +version = "2.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bee3b67d0ea5c2cca5003417989af8996f8604e34fb9ddf96208a033901e70de" +checksum = "7e7ca276ad3145661a65914e6daf131ca5120cd3dcee8f8f3214b8875184a148" dependencies = [ + "displaydoc", "icu_collections", "icu_locale_core", "icu_properties_data", @@ -3732,15 +3746,15 @@ dependencies = [ [[package]] name = "icu_properties_data" -version = "2.2.0" +version = "2.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8e2bbb201e0c04f7b4b3e14382af113e17ba4f63e2c9d2ee626b720cbce54a14" +checksum = "e590f038c1464a96894fd6d10127e90a8be4509f56ff7ecef851b15cee0b7caa" [[package]] name = "icu_provider" -version = "2.2.0" +version = "2.3.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "139c4cf31c8b5f33d7e199446eff9c1e02decfc2f0eec2c8d71f65befa45b421" +checksum = "d27bbb9d3abbefac45d55f647c9de1d44aafcd1186eb91879afef17c396c3e73" dependencies = [ "displaydoc", "icu_locale_core", @@ -3751,12 +3765,6 @@ dependencies = [ "zerovec", ] -[[package]] -name = "id-arena" -version = "2.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3d3067d79b975e8844ca9eb072e16b31c3c1c36928edf9c6789548c524d0d954" - [[package]] name = "ident_case" version = "1.0.1" @@ -3786,24 +3794,12 @@ dependencies = [ [[package]] name = "indexmap" -version = "1.9.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bd070e393353796e801d209ad339e89596eb4c8d430d18ede6a1cced8fafbd99" -dependencies = [ - "autocfg", - "hashbrown 0.12.3", -] - -[[package]] -name = "indexmap" -version = "2.14.0" +version = "2.14.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d466e9454f08e4a911e14806c24e16fba1b4c121d1ea474396f396069cf949d9" +checksum = "07aa2048142242915a31d35844fb311e0e53fcca590c3a0a40dcf1b841fa09eb" dependencies = [ "equivalent", "hashbrown 0.17.1", - "serde", - "serde_core", ] [[package]] @@ -3822,7 +3818,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "232929e1d75fe899576a3d5c7416ad0d88dbfbb3c3d6aa00873a7408a50ddb88" dependencies = [ "ahash", - "indexmap 2.14.0", + "indexmap", "is-terminal", "itoa", "log", @@ -3835,9 +3831,9 @@ dependencies = [ [[package]] name = "insta" -version = "1.47.2" +version = "1.48.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7b4a6248eb93a4401ed2f37dfe8ea592d3cf05b7cf4f8efa867b6895af7e094e" +checksum = "86f0f8fee8c926415c58d6ae43a08523a26faccb2323f5e6b644fe7dd4ef6b82" dependencies = [ "console", "once_cell", @@ -3856,20 +3852,20 @@ dependencies = [ [[package]] name = "io-uring" -version = "0.7.12" +version = "0.7.14" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4d09b98f7eace8982db770e4408e7470b028ce513ac28fecdc6bf4c30fe92b62" +checksum = "d64d8ca234d152948ceaede1f419b6a83983a5ecccaac05fb337a809c96d3aa6" dependencies = [ - "bitflags 2.11.1", + "bitflags 2.13.1", "cfg-if", "libc", ] [[package]] name = "ipnet" -version = "2.12.0" +version = "2.12.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d98f6fed1fde3f8c21bc40a1abb88dd75e67924f9cffc3ef95607bad8017f8e2" +checksum = "6a756c3fac73139e83f14c2d742155dd2b78d3ee56597b419a0579b7bdd6dd78" [[package]] name = "is-terminal" @@ -3914,35 +3910,47 @@ checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682" [[package]] name = "jiff" -version = "0.2.24" +version = "0.2.35" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f00b5dbd620d61dfdcb6007c9c1f6054ebd75319f163d886a9055cec1155073d" +checksum = "668b7183bd07af9a4885f5c35b0cc5c83c4607a913c16b7e17291832910d2dcc" dependencies = [ + "defmt", + "jiff-core", "jiff-static", "jiff-tzdb-platform", "log", "portable-atomic", "portable-atomic-util", "serde_core", - "windows-sys 0.61.2", + "windows-link", +] + +[[package]] +name = "jiff-core" +version = "0.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7feca88439efe53da3754500c1851dedf3cb36c524dd5cf8225cc0794de95d09" +dependencies = [ + "defmt", ] [[package]] name = "jiff-static" -version = "0.2.24" +version = "0.2.35" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e000de030ff8022ea1da3f466fbb0f3a809f5e51ed31f6dd931c35181ad8e6d7" +checksum = "3a69dcb3a21cfb32ce1cd056169337ca284af0766dd766e7878819b251a49204" dependencies = [ + "jiff-core", "proc-macro2", "quote", - "syn 2.0.117", + "syn 2.0.119", ] [[package]] name = "jiff-tzdb" -version = "0.1.6" +version = "0.1.8" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c900ef84826f1338a557697dc8fc601df9ca9af4ac137c7fb61d4c6f2dfd3076" +checksum = "142bd39932ad231f10513df9ab62661fead8719872150b7ad02a2df79f4e141e" [[package]] name = "jiff-tzdb-platform" @@ -3994,28 +4002,27 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "38c0b942f458fe50cdac086d2f946512305e5631e720728f2a61aabcd47a6264" dependencies = [ "quote", - "syn 2.0.117", + "syn 2.0.119", ] [[package]] name = "jobserver" -version = "0.1.34" +version = "0.1.35" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9afb3de4395d6b3e67a780b6de64b51c978ecf11cb9a462c66be7d4ca9039d33" +checksum = "1c00acbd29eabad4a2392fa0e921c874934dbbf4194312ad20f04a0ed67a3cb3" dependencies = [ - "getrandom 0.3.4", + "getrandom 0.4.3", "libc", ] [[package]] name = "js-sys" -version = "0.3.95" +version = "0.3.103" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2964e92d1d9dc3364cae4d718d93f227e3abb088e747d92e0395bfdedf1c12ca" +checksum = "53b44bfcdb3f8d5837a46dae1ca9660a837176eee74a28b229bc626816589102" dependencies = [ "cfg-if", "futures-util", - "once_cell", "wasm-bindgen", ] @@ -4025,7 +4032,7 @@ version = "0.7.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "b750dcadc39a09dbadd74e118f6dd6598df77fa01df0cfcdc52c28dece74528a" dependencies = [ - "bitflags 2.11.1", + "bitflags 2.13.1", "serde", ] @@ -4046,9 +4053,9 @@ checksum = "a4933f3f57a8e9d9da04db23fb153356ecaf00cbd14aee46279c33dc80925c37" [[package]] name = "lazy-js-bundle" -version = "0.7.6" +version = "0.7.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ebbde2c5796719fbd82d6b8ec0be3dacf1f70c2876dee0f2c001632794d6641f" +checksum = "51cdaa5abc885e2d606e6985281ace220778bde190be20aab1ec2346ed8bd1fa" [[package]] name = "lazy_static" @@ -4056,12 +4063,6 @@ version = "1.5.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "bbd2bcb4c963f2ddae06a2efc7e9f3591312473c50c6685e1f298068316e66fe" -[[package]] -name = "leb128fmt" -version = "0.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "09edd9e8b54e49e587e4f6295a7d29c3ea94d469cb40ab8ca70b288248a81db2" - [[package]] name = "lexical-core" version = "1.0.6" @@ -4121,21 +4122,21 @@ dependencies = [ [[package]] name = "libbz2-rs-sys" -version = "0.2.3" +version = "0.2.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b3a6a8c165077efc8f3a971534c50ea6a1a18b329ef4a66e897a7e3a1494565f" +checksum = "34b357333733e8260735ba5894eb928c02ecc69c78715f01a8019e7fa7f2db4c" [[package]] name = "libc" -version = "0.2.186" +version = "0.2.189" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "68ab91017fe16c622486840e4c83c9a37afeff978bd239b5293d61ece587de66" +checksum = "3eaf3ede3fee6db1a4c2ee091bf8a8b4dccdc6d17f656fb07896ee72867612f2" [[package]] name = "libfuzzer-sys" -version = "0.4.12" +version = "0.4.13" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f12a681b7dd8ce12bff52488013ba614b869148d54dd79836ab85aafdd53f08d" +checksum = "a9fd2f41a1cba099f79a0b6b6c35656cf7c03351a7bae8ff0f28f25270f929d2" dependencies = [ "arbitrary", "cc", @@ -4153,18 +4154,18 @@ dependencies = [ [[package]] name = "liblzma" -version = "0.4.6" +version = "0.4.8" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b6033b77c21d1f56deeae8014eb9fbe7bdf1765185a6c508b5ca82eeaed7f899" +checksum = "2fe0a34ca854fd4f20c07f696fc8675aec78f87d88d29f5e10257a7490a1b2e1" dependencies = [ "liblzma-sys", ] [[package]] name = "liblzma-sys" -version = "0.4.6" +version = "0.4.8" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1a60851d15cd8c5346eca4ab8babff585be2ae4bc8097c067291d3ffe2add3b6" +checksum = "a0dad045e4b1b7b170be4b60b54b780cafb4490165461bac7d1cf7b703f61d5f" dependencies = [ "cc", "libc", @@ -4222,7 +4223,7 @@ dependencies = [ "object_store", "parquet", "parquet-variant-compute", - "rand 0.10.1", + "rand 0.10.2", "serde", "serde_json", "shuttle", @@ -4287,7 +4288,7 @@ dependencies = [ "prost", "serde", "tempfile", - "thiserror 2.0.18", + "thiserror 2.0.20", "tokio", "url", ] @@ -4301,6 +4302,7 @@ dependencies = [ "arrow-schema", "bytes", "datafusion", + "datafusion-datasource", "divan", "fastrace", "futures", @@ -4310,7 +4312,7 @@ dependencies = [ "object_store", "parquet", "parquet-variant-json", - "rand 0.10.1", + "rand 0.10.2", "serde_json", "shuttle", "t4", @@ -4385,7 +4387,7 @@ dependencies = [ "tempfile", "tokio", "tonic", - "tower-http", + "tower-http 0.7.1", "url", "uuid", ] @@ -4402,9 +4404,9 @@ dependencies = [ [[package]] name = "litemap" -version = "0.8.2" +version = "0.8.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "92daf443525c4cce67b150400bc2316076100ce0b3686209eb8cf3c31612e6f0" +checksum = "47d9d19d1d6efa0109d2f65ff4c85cddd50bd572e5a00127ab10987290bcefae" [[package]] name = "litrs" @@ -4423,44 +4425,40 @@ dependencies = [ [[package]] name = "log" -version = "0.4.32" +version = "0.4.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "953f07c43838f8e6f9758cab68bf5bed85465e7587ebe0b823f1bcd81978ad3a" -dependencies = [ - "sval", - "sval_ref", - "value-bag", -] +checksum = "f9f8bd3e56ce4dfc153cf470fffbfa98c7620958b312ca5c3a4b8d5181fd13c6" [[package]] name = "logforth" -version = "0.29.1" +version = "0.30.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "40c105c59828d07aeb95b06f9a345b12869ddc249d44a7302697a66da439076f" +checksum = "522000d3921e4b089de59204d2d3ca8792cd53f8ce5a54b5ac8b9a6e867259f0" dependencies = [ - "logforth-append-opentelemetry", + "log", + "logforth-append-file", "logforth-bridge-log", "logforth-core", + "logforth-filter-rustlog", + "logforth-layout-json", + "logforth-layout-text", ] [[package]] -name = "logforth-append-opentelemetry" -version = "0.3.1" +name = "logforth-append-file" +version = "0.4.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "544f950997c23b8b0a8c324af05aad29db446738fb04ae57cbcdf233d67b1865" +checksum = "80f4ed78a03a30c12285135d98330f0f5d53cb0019bd1ac6d66863dae72d8d52" dependencies = [ + "jiff", "logforth-core", - "logforth-layout-json", - "opentelemetry", - "opentelemetry-otlp", - "opentelemetry_sdk", ] [[package]] name = "logforth-bridge-log" -version = "0.3.0" +version = "0.4.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a4aa6ca548389fd166a995b5940e15b0dacbdd5a30f2f24eac9aa4bf664bda5c" +checksum = "9c7224c78547e542572ae4d1f787f96c4395a4c3f24513bf221bd8e49888f610" dependencies = [ "log", "logforth-core", @@ -4468,19 +4466,28 @@ dependencies = [ [[package]] name = "logforth-core" -version = "0.3.1" +version = "0.4.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a77869b8dba38c67ed19e1753e59d9faefdcc60557bc4e84db0348606a304ac5" +checksum = "400ba5305e0cb6819efa6c2c503020562eb5b47fd53c91c935be55af1ffd374e" dependencies = [ "anyhow", - "value-bag", + "serde", +] + +[[package]] +name = "logforth-filter-rustlog" +version = "0.4.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5e8431cfa1d3eeca479c363651320a66c7003350528f678b30e8c1474fb0d802" +dependencies = [ + "logforth-core", ] [[package]] name = "logforth-layout-json" -version = "0.3.0" +version = "0.4.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "01b80d310e0670560404a825f64dbd78a8761c5bb7da952513e90ba9dd525bd2" +checksum = "9d86a105f8f32151ca1e0a6d4097d478badb02d5715239ba0fa431a188a5d966" dependencies = [ "jiff", "logforth-core", @@ -4488,6 +4495,17 @@ dependencies = [ "serde_json", ] +[[package]] +name = "logforth-layout-text" +version = "0.4.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4eafe007ac55293d807e8c7f5a23002e2518d32e476e195443cde23e7845416a" +dependencies = [ + "colored", + "jiff", + "logforth-core", +] + [[package]] name = "longest-increasing-subsequence" version = "0.1.0" @@ -4526,14 +4544,14 @@ checksum = "1b27834086c65ec3f9387b096d66e99f221cf081c2b738042aa252bcd41204e3" dependencies = [ "proc-macro2", "quote", - "syn 2.0.117", + "syn 2.0.119", ] [[package]] name = "manganis" -version = "0.7.6" +version = "0.7.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2e06225f29a781d86afdfafa562de09621f2ace377136ae2ae9ca9e72a29b920" +checksum = "1ed2af894a724a0b6e521f9efe84f98cfdc74a4de6828efef77ffd3b1679fd52" dependencies = [ "const-serialize 0.7.2", "const-serialize 0.8.0-alpha.0", @@ -4542,14 +4560,14 @@ dependencies = [ "manganis-macro", "ndk-context", "objc2", - "thiserror 2.0.18", + "thiserror 2.0.20", ] [[package]] name = "manganis-core" -version = "0.7.6" +version = "0.7.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "774ddc382b4fb30f3fdcf2418131cbac5e6111f00e6c7ddf9e598ef3b2b4cd91" +checksum = "e258c71b137bf48deaffb48b8a8a2f994af9e5c3ca201daa20d41e26b379da11" dependencies = [ "const-serialize 0.7.2", "const-serialize 0.8.0-alpha.0", @@ -4561,16 +4579,16 @@ dependencies = [ [[package]] name = "manganis-macro" -version = "0.7.6" +version = "0.7.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "731c83c89d831f341fb46eba0aefdfb3433f77a3f78a8b08d9d88746613a8f8b" +checksum = "0abae8346bbf637e6d21ac3831863bd5a98657517f5cd2d69d651046b95e7ce9" dependencies = [ "dunce", "macro-string", "manganis-core", "proc-macro2", "quote", - "syn 2.0.117", + "syn 2.0.119", ] [[package]] @@ -4625,9 +4643,9 @@ dependencies = [ [[package]] name = "memmap2" -version = "0.9.10" +version = "0.9.11" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "714098028fe011992e1c3962653c96b2d578c4b4bce9036e15ff220319b1e0e3" +checksum = "d1219ed1b7f229ee7104d281dd01d6802fe28bb6e95d292942c4daacdeb798c0" dependencies = [ "libc", ] @@ -4667,16 +4685,19 @@ dependencies = [ ] [[package]] -name = "minimal-lexical" -version = "0.2.1" +name = "miniz_oxide" +version = "0.8.9" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "68354c5c6bd36d73ff3feceb05efa59b6acb7626617f4962be322a825e61f79a" +checksum = "1fa76a2c86f704bdb222d66965fb3d63269ce38518b83cb0575fca855ebb6316" +dependencies = [ + "adler2", +] [[package]] name = "miniz_oxide" -version = "0.8.9" +version = "0.9.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1fa76a2c86f704bdb222d66965fb3d63269ce38518b83cb0575fca855ebb6316" +checksum = "b63fbc4a50860e98e7b2aa7804ded1db5cbc3aff9193adaff57a6931bf7c4b4c" dependencies = [ "adler2", "simd-adler32", @@ -4684,9 +4705,9 @@ dependencies = [ [[package]] name = "mio" -version = "1.2.0" +version = "1.2.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "50b7e5b27aa02a74bac8c3f23f448f8d87ff11f92d3aac1a6ed369ee08cc56c1" +checksum = "30d65c71f1ce40ab09135ce117d742b9f8a19ff91a41a8b57ed50bc2de59c427" dependencies = [ "libc", "wasi", @@ -4706,7 +4727,7 @@ dependencies = [ "httparse", "memchr", "mime", - "spin 0.9.8", + "spin 0.9.9", "version_check", ] @@ -4716,7 +4737,7 @@ version = "0.9.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "c3f42e7bbe13d351b6bead8286a43aac9534b82bd3cc43e47037f012ebfd62d4" dependencies = [ - "bitflags 2.11.1", + "bitflags 2.13.1", "jni-sys 0.3.1", "log", "ndk-sys", @@ -4753,12 +4774,11 @@ dependencies = [ [[package]] name = "nom" -version = "7.1.3" +version = "8.0.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d273983c5a657a70a3e8f2a01329822f3b8c8172b73826411a55751e404a0a4a" +checksum = "df9761775871bdef83bee530e60050f7e54b1105350d6884eb0fb4f46c2f9405" dependencies = [ "memchr", - "minimal-lexical", ] [[package]] @@ -4781,9 +4801,9 @@ dependencies = [ [[package]] name = "num-bigint" -version = "0.4.6" +version = "0.4.8" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a5e44f723f1133c9deac646763579fdb3ac745e418f2a7af9cd0c431da1f20b9" +checksum = "c89e69e7e0f03bea5ef08013795c25018e101932225a656383bd384495ecc367" dependencies = [ "num-integer", "num-traits", @@ -4810,9 +4830,9 @@ dependencies = [ [[package]] name = "num-conv" -version = "0.2.1" +version = "0.2.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c6673768db2d862beb9b39a78fdcb1a69439615d5794a1be50caa9bc92c81967" +checksum = "521739c6d2bac4aa25192232afe6841231376b2b26d4d9fae5ecf8ca5772e441" [[package]] name = "num-format" @@ -4826,9 +4846,9 @@ dependencies = [ [[package]] name = "num-integer" -version = "0.1.46" +version = "0.1.47" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7969661fd2958a5cb096e56c8e1ad0444ac2bbcd0061bd28660485a44879858f" +checksum = "7ce2d95d4b3734dc35aa2f45e1aa22cd416814592a4f9d9205e11affd5b8e10b" dependencies = [ "num-traits", ] @@ -4862,7 +4882,7 @@ dependencies = [ "proc-macro-crate", "proc-macro2", "quote", - "syn 2.0.117", + "syn 2.0.119", ] [[package]] @@ -4880,7 +4900,9 @@ version = "0.3.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "2a180dd8642fa45cdb7dd721cd4c11b1cadd4929ce112ebd8b9f5803cc79d536" dependencies = [ - "bitflags 2.11.1", + "bitflags 2.13.1", + "dispatch2", + "objc2", ] [[package]] @@ -4889,6 +4911,16 @@ version = "4.1.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "ef25abbcd74fb2609453eb695bd2f860d389e457f67dc17cafc8b8cbc89d0c33" +[[package]] +name = "objc2-foundation" +version = "0.3.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e3e0adef53c21f888deb4fa59fc59f7eb17404926ee8a6f59f5df0fd7f9f3272" +dependencies = [ + "bitflags 2.13.1", + "objc2", +] + [[package]] name = "objc2-io-kit" version = "0.3.2" @@ -4899,6 +4931,17 @@ dependencies = [ "objc2-core-foundation", ] +[[package]] +name = "objc2-open-directory" +version = "0.3.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bb82bed227edf5201dfedf072bba4015a33d3d4a98519837295a90f0a23f676d" +dependencies = [ + "objc2", + "objc2-core-foundation", + "objc2-foundation", +] + [[package]] name = "object" version = "0.37.3" @@ -4908,6 +4951,15 @@ dependencies = [ "memchr", ] +[[package]] +name = "object" +version = "0.39.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2e5a6c098c7a3b6547378093f5cc30bc54fd361ce711e05293a5cc589562739b" +dependencies = [ + "memchr", +] + [[package]] name = "object_store" version = "0.13.2" @@ -4930,14 +4982,14 @@ dependencies = [ "md-5 0.10.6", "parking_lot", "percent-encoding", - "quick-xml 0.39.2", - "rand 0.10.1", + "quick-xml 0.39.4", + "rand 0.10.2", "reqwest 0.12.28", "ring", "serde", "serde_json", "serde_urlencoded", - "thiserror 2.0.18", + "thiserror 2.0.20", "tokio", "tracing", "url", @@ -4966,36 +5018,36 @@ checksum = "7c87def4c32ab89d880effc9e097653c8da5d6ef28e6b539d313baaacfbafcbe" [[package]] name = "opentelemetry" -version = "0.31.0" +version = "0.32.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b84bcd6ae87133e903af7ef497404dda70c60d0ea14895fc8a5e6722754fc2a0" +checksum = "b0142c63252a9e054e68a4c61a5778f7b14f576274d593f8ce883d191a099682" dependencies = [ "futures-core", "futures-sink", "js-sys", "pin-project-lite", - "thiserror 2.0.18", + "thiserror 2.0.20", "tracing", ] [[package]] name = "opentelemetry-http" -version = "0.31.0" +version = "0.32.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d7a6d09a73194e6b66df7c8f1b680f156d916a1a942abf2de06823dd02b7855d" +checksum = "5683015d09e2df236ef005b17f6f196f0d5f6313c4fa43a7b6a53b52776e4331" dependencies = [ "async-trait", "bytes", "http", "opentelemetry", - "reqwest 0.12.28", + "reqwest 0.13.4", ] [[package]] name = "opentelemetry-otlp" -version = "0.31.1" +version = "0.32.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1f69cd6acbb9af919df949cd1ec9e5e7fdc2ef15d234b6b795aaa525cc02f71f" +checksum = "9966929966d17620d7c316c643ba62631826e10021409357772d5eea84f62c35" dependencies = [ "http", "opentelemetry", @@ -5003,18 +5055,18 @@ dependencies = [ "opentelemetry-proto", "opentelemetry_sdk", "prost", - "reqwest 0.12.28", - "thiserror 2.0.18", + "reqwest 0.13.4", + "thiserror 2.0.20", "tokio", "tonic", - "tracing", + "tonic-types", ] [[package]] name = "opentelemetry-proto" -version = "0.31.0" +version = "0.32.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a7175df06de5eaee9909d4805a3d07e28bb752c34cab57fa9cff549da596b30f" +checksum = "56d658ba1faf63f7b9c492cfbe6e0ec365440a16132d3270c1065f7b33f1b638" dependencies = [ "opentelemetry", "opentelemetry_sdk", @@ -5025,24 +5077,25 @@ dependencies = [ [[package]] name = "opentelemetry_sdk" -version = "0.31.0" +version = "0.32.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e14ae4f5991976fd48df6d843de219ca6d31b01daaab2dad5af2badeded372bd" +checksum = "9b59f80e1ac4d5ff7a2db8fb6c80badb7f0f3f858211fba08dd9aaec750894f9" dependencies = [ "futures-channel", "futures-executor", "futures-util", "opentelemetry", "percent-encoding", - "rand 0.9.4", - "thiserror 2.0.18", + "portable-atomic", + "rand 0.9.5", + "thiserror 2.0.20", ] [[package]] name = "owo-colors" -version = "3.5.0" +version = "4.4.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c1b04fb49957986fdce4d6ee7a65027d55d4b6d2265e5848bbb507b58ccfdb6f" +checksum = "13c45bb4a6ae1280ec0803b1ef9d3455eb50f01efbbe1447ab020f1d54fba9d8" [[package]] name = "parking_lot" @@ -5114,7 +5167,7 @@ dependencies = [ "arrow-schema", "chrono", "half", - "indexmap 2.14.0", + "indexmap", "num-traits", "simdutf8", "uuid", @@ -5130,7 +5183,7 @@ dependencies = [ "arrow-schema", "chrono", "half", - "indexmap 2.14.0", + "indexmap", "parquet-variant", "parquet-variant-json", "serde_json", @@ -5152,10 +5205,10 @@ dependencies = [ ] [[package]] -name = "paste" -version = "1.0.15" +name = "pastey" +version = "0.2.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "57c0d7b74b563b49d38dae00a0c37d4d6de9b432382b2892f0574ddcae73fd0a" +checksum = "2ee67f1008b1ba2321834326597b8e186293b049a023cdef258527550b9935b4" [[package]] name = "percent-encoding" @@ -5169,7 +5222,7 @@ version = "0.1.8" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "575828d9d7d205188048eb1508560607a03d21eafdbba47b8cade1736c1c28e1" dependencies = [ - "bitflags 2.11.1", + "bitflags 2.13.1", "c-enum", "perf-event-open-sys2", ] @@ -5190,7 +5243,7 @@ version = "0.7.4" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "0939b8fad77dfaeb29ebbd35faaeaadbf833167f30975f1b8993bbba09ea0a0f" dependencies = [ - "bitflags 2.11.1", + "bitflags 2.13.1", "c-enum", "libc", "memmap2", @@ -5200,9 +5253,9 @@ dependencies = [ [[package]] name = "pest" -version = "2.8.6" +version = "2.9.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e0848c601009d37dfa3430c4666e147e49cdcf1b92ecd3e63657d8a5f19da662" +checksum = "5a07a60cc7a4d00c91f95c685609d1d2f79050e6804b70ebedd7650f0b839bcf" dependencies = [ "memchr", "ucd-trie", @@ -5210,9 +5263,9 @@ dependencies = [ [[package]] name = "pest_derive" -version = "2.8.6" +version = "2.9.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "11f486f1ea21e6c10ed15d5a7c77165d0ee443402f0780849d1768e7d9d6fe77" +checksum = "b3a83744a5c8455b8b3e0dc5031362780a347c878bdd11584d1a8984228cc88d" dependencies = [ "pest", "pest_generator", @@ -5220,25 +5273,24 @@ dependencies = [ [[package]] name = "pest_generator" -version = "2.8.6" +version = "2.9.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8040c4647b13b210a963c1ed407c1ff4fdfa01c31d6d2a098218702e6664f94f" +checksum = "e0cd3451aa3de60d4b9a1e736885e4dea6b31617598026f12256ad566d63304a" dependencies = [ "pest", "pest_meta", "proc-macro2", "quote", - "syn 2.0.117", + "syn 2.0.119", ] [[package]] name = "pest_meta" -version = "2.8.6" +version = "2.9.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "89815c69d36021a140146f26659a81d6c2afa33d216d736dd4be5381a7362220" +checksum = "e04d3a0849e241d7dfce834c83b1c5edc8622009e8dd51a12ba1927c32f05496" dependencies = [ "pest", - "sha2 0.10.9", ] [[package]] @@ -5249,7 +5301,7 @@ checksum = "8701b58ea97060d5e5b155d383a69952a60943f0e6dfe30b04c287beb0b27455" dependencies = [ "fixedbitset", "hashbrown 0.15.5", - "indexmap 2.14.0", + "indexmap", "serde", ] @@ -5273,22 +5325,22 @@ dependencies = [ [[package]] name = "pin-project" -version = "1.1.11" +version = "1.1.13" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f1749c7ed4bcaf4c3d0a3efc28538844fb29bcdd7d2b67b2be7e20ba861ff517" +checksum = "2466b2336ed02bcdca6b294417127b90ec92038d1d5c4fbeac971a922e0e0924" dependencies = [ "pin-project-internal", ] [[package]] name = "pin-project-internal" -version = "1.1.11" +version = "1.1.13" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d9b20ed30f105399776b9c883e68e536ef602a16ae6f596d2c473591d6ad64c6" +checksum = "c96395f0a926bc13b1c17622aaddda1ecb55d49c8f1bf9777e4d877800a43f8b" dependencies = [ "proc-macro2", "quote", - "syn 2.0.117", + "syn 2.0.119", ] [[package]] @@ -5299,9 +5351,9 @@ checksum = "a89322df9ebe1c1578d689c92318e070967d1042b512afbe49518723f4e6d5cd" [[package]] name = "pkg-config" -version = "0.3.33" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "19f132c84eca552bf34cab8ec81f1c1dcc229b811638f9d283dceabe58c5569e" +checksum = "f6b464fbc74e149a392436b17d523f769e057cb6877f6a5c4618bc6f11800548" [[package]] name = "plain" @@ -5317,9 +5369,9 @@ checksum = "2f3a9f18d041e6d0e102a0a46750538147e5e8992d3b4873aaafee2520b00ce3" [[package]] name = "portable-atomic" -version = "1.13.1" +version = "1.15.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c33a9471896f1c69cecef8d20cbe2f7accd12527ce60845ff44c153bb2a21b49" +checksum = "05c8b63e8d9609db387f0324918f81d68fe27748f084ef092fb35954d0539a85" [[package]] name = "portable-atomic-util" @@ -5332,9 +5384,9 @@ dependencies = [ [[package]] name = "potential_utf" -version = "0.1.5" +version = "0.1.6" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0103b1cef7ec0cf76490e969665504990193874ea05c85ff9bab8b911d0a0564" +checksum = "d83eb9bc6d8e5cf568e7a1101d60ee05e81ed50ea106026f3d18deeb046d7661" dependencies = [ "zerovec", ] @@ -5361,10 +5413,10 @@ dependencies = [ "nix", "once_cell", "smallvec", - "spin 0.10.0", + "spin 0.10.1", "symbolic-demangle", "tempfile", - "thiserror 2.0.18", + "thiserror 2.0.20", ] [[package]] @@ -5382,16 +5434,6 @@ version = "0.4.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "9a65843dfefbafd3c879c683306959a6de478443ffe9c9adf02f5976432402d7" -[[package]] -name = "prettyplease" -version = "0.2.37" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "479ca8adacdd7ce8f1fb39ce9ecccbfe93a3f1344b3d0d97f20bc0196208f62b" -dependencies = [ - "proc-macro2", - "syn 2.0.117", -] - [[package]] name = "proc-macro-crate" version = "3.5.0" @@ -5401,33 +5443,11 @@ dependencies = [ "toml_edit", ] -[[package]] -name = "proc-macro-error-attr2" -version = "2.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "96de42df36bb9bba5542fe9f1a054b8cc87e172759a1868aa05c1f3acc89dfc5" -dependencies = [ - "proc-macro2", - "quote", -] - -[[package]] -name = "proc-macro-error2" -version = "2.0.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "11ec05c52be0a07b08061f7dd003e7d7092e0472bc731b4af7bb1ef876109802" -dependencies = [ - "proc-macro-error-attr2", - "proc-macro2", - "quote", - "syn 2.0.117", -] - [[package]] name = "proc-macro2" -version = "1.0.106" +version = "1.0.107" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8fd00f0bb2e90d81d1044c2b32617f68fcb9fa3bb7640c23e9c748e53fb30934" +checksum = "985e7ec9bb745e6ce6535b544d84d6cd6f7ad8bd711c398938ae983b91a766d9" dependencies = [ "unicode-ident", ] @@ -5440,7 +5460,7 @@ checksum = "af066a9c399a26e020ada66a034357a868728e72cd426f3adcd35f80d88d88c8" dependencies = [ "proc-macro2", "quote", - "syn 2.0.117", + "syn 2.0.119", "version_check", ] @@ -5464,14 +5484,14 @@ dependencies = [ "itertools 0.14.0", "proc-macro2", "quote", - "syn 2.0.117", + "syn 2.0.119", ] [[package]] name = "prost-types" -version = "0.14.3" +version = "0.14.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8991c4cbdb8bc5b11f0b074ffe286c30e523de90fee5ba8132f1399f23cb3dd7" +checksum = "f94967dc7688f3054c7fac87473ffae4cc4c3904800e2d9f5b857246d8963b0a" dependencies = [ "prost", ] @@ -5484,9 +5504,9 @@ checksum = "33cb294fe86a74cbcf50d4445b37da762029549ebeea341421c7c70370f86cac" [[package]] name = "psm" -version = "0.1.31" +version = "0.1.32" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "645dbe486e346d9b5de3ef16ede18c26e6c70ad97418f4874b8b1889d6e761ea" +checksum = "4dcd034599e63b970727f70d79e02d62390a4a84f7c6b827c27c46d5ac3fa622" dependencies = [ "ar_archive_writer", "cc", @@ -5513,9 +5533,9 @@ dependencies = [ [[package]] name = "quick-xml" -version = "0.39.2" +version = "0.39.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "958f21e8e7ceb5a1aa7fa87fab28e7c75976e0bfe7e23ff069e0a260f894067d" +checksum = "cdcc8dd4e2f670d309a5f0e83fe36dfdc05af317008fea29144da1a2ac858e5e" dependencies = [ "memchr", "serde", @@ -5523,19 +5543,19 @@ dependencies = [ [[package]] name = "quinn" -version = "0.11.9" +version = "0.11.11" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b9e20a958963c291dc322d98411f541009df2ced7b5a4f2bd52337638cfccf20" +checksum = "0c1a41e437b6bbd489372cd4971de128e85c855f56c57f283d20ff016cf7c0a8" dependencies = [ "bytes", "cfg_aliases", "pin-project-lite", "quinn-proto", "quinn-udp", - "rustc-hash 2.1.2", + "rustc-hash 2.1.3", "rustls", "socket2", - "thiserror 2.0.18", + "thiserror 2.0.20", "tokio", "tracing", "web-time", @@ -5543,20 +5563,21 @@ dependencies = [ [[package]] name = "quinn-proto" -version = "0.11.14" +version = "0.11.17" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "434b42fec591c96ef50e21e886936e66d3cc3f737104fdb9b737c40ffb94c098" +checksum = "04759210543be93709136e28212294a659ef5001836ff4eab4d663e4529bba83" dependencies = [ "bytes", - "getrandom 0.3.4", + "getrandom 0.4.3", "lru-slab", - "rand 0.9.4", + "rand 0.10.2", + "rand_pcg 0.10.2", "ring", - "rustc-hash 2.1.2", + "rustc-hash 2.1.3", "rustls", "rustls-pki-types", "slab", - "thiserror 2.0.18", + "thiserror 2.0.20", "tinyvec", "tracing", "web-time", @@ -5564,23 +5585,23 @@ dependencies = [ [[package]] name = "quinn-udp" -version = "0.5.14" +version = "0.5.15" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "addec6a0dcad8a8d96a771f815f0eaf55f9d1805756410b39f5fa81332574cbd" +checksum = "35a133f956daabe89a61a685c2649f13d82d5aa4bd5d12d1277e1072a21c0694" dependencies = [ "cfg_aliases", "libc", "once_cell", "socket2", "tracing", - "windows-sys 0.59.0", + "windows-sys 0.61.2", ] [[package]] name = "quote" -version = "1.0.45" +version = "1.0.47" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "41f2619966050689382d2b44f664f4bc593e129785a36d6ee376ddf37259b924" +checksum = "1fbf4db142a473a8d80c26bbf18454ed458bf8d26c8219c331daecfdbd079001" dependencies = [ "proc-macro2", ] @@ -5605,9 +5626,9 @@ checksum = "dc33ff2d4973d518d823d61aa239014831e521c75da58e3df4840d3f47749d09" [[package]] name = "rand" -version = "0.8.6" +version = "0.8.8" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5ca0ecfa931c29007047d1bc58e623ab12e5590e8c7cc53200d5202b69266d8a" +checksum = "e058c7de0b26af77780c769414d6257830bb240f3c38477dbc2c16e5f54d6d4c" dependencies = [ "libc", "rand_chacha 0.3.1", @@ -5616,9 +5637,9 @@ dependencies = [ [[package]] name = "rand" -version = "0.9.4" +version = "0.9.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "44c5af06bb1b7d3216d91932aed5265164bf384dc89cd6ba05cf59a35f5f76ea" +checksum = "b9ef1d0d795eb7d84685bca4f72f3649f064e6641543d3a8c415898726a57b41" dependencies = [ "rand_chacha 0.9.0", "rand_core 0.9.5", @@ -5626,12 +5647,12 @@ dependencies = [ [[package]] name = "rand" -version = "0.10.1" +version = "0.10.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d2e8e8bcc7961af1fdac401278c6a831614941f6164ee3bf4ce61b7edb162207" +checksum = "c7f5fa3a058cd35567ef9bfa5e75732bee0f9e4c55fa90477bef2dfcdbc4be80" dependencies = [ "chacha20", - "getrandom 0.4.2", + "getrandom 0.4.3", "rand_core 0.10.1", ] @@ -5688,6 +5709,15 @@ dependencies = [ "rand_core 0.6.4", ] +[[package]] +name = "rand_pcg" +version = "0.10.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "caa0f4137e1c0a72f4c651489402276c8e8e1cf081f3b0ba156d2cbeef09e86a" +dependencies = [ + "rand_core 0.10.1", +] + [[package]] name = "raw-window-handle" version = "0.6.2" @@ -5711,7 +5741,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "76009fbe0614077fc1a2ce255e3a1881a2e3a3527097d5dc6d8212c585e7e38b" dependencies = [ "quote", - "syn 2.0.117", + "syn 2.0.119", ] [[package]] @@ -5720,14 +5750,14 @@ version = "0.5.18" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "ed2bf2547551a7053d6fdfafda3f938979645c44812fbfcda098faae3f1a362d" dependencies = [ - "bitflags 2.11.1", + "bitflags 2.13.1", ] [[package]] name = "regex" -version = "1.12.4" +version = "1.13.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f1292b7759ae1cb9ec195452d1390a074f0cd8541ab7a5a8c31cd6db45d4a6ba" +checksum = "f020237b6c8eed93db2e2cb53c00c60a8e1bc73da7d073199a1180401450218d" dependencies = [ "aho-corasick", "memchr", @@ -5737,9 +5767,9 @@ dependencies = [ [[package]] name = "regex-automata" -version = "0.4.14" +version = "0.4.18" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6e1dd4122fc1595e8162618945476892eefca7b88c52820e74af6262213cae8f" +checksum = "ad8553b9b26413251cbf30e620595c7a41b3887f03da04579c0e6b0d6a06b4b2" dependencies = [ "aho-corasick", "memchr", @@ -5768,7 +5798,6 @@ dependencies = [ "bytes", "cookie", "cookie_store", - "futures-channel", "futures-core", "futures-util", "h2", @@ -5795,7 +5824,7 @@ dependencies = [ "tokio-rustls", "tokio-util", "tower", - "tower-http", + "tower-http 0.6.11", "tower-service", "url", "wasm-bindgen", @@ -5813,7 +5842,9 @@ checksum = "219c5811de6525e5416c7d5d53bb656d3afdbc6c5af816e0802bcfa42dbdc1c3" dependencies = [ "base64 0.22.1", "bytes", + "futures-channel", "futures-core", + "futures-util", "http", "http-body", "http-body-util", @@ -5828,7 +5859,7 @@ dependencies = [ "sync_wrapper", "tokio", "tower", - "tower-http", + "tower-http 0.6.11", "tower-service", "url", "wasm-bindgen", @@ -5861,15 +5892,15 @@ dependencies = [ [[package]] name = "rtrb" -version = "0.3.4" +version = "0.3.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4ade083ccbb4bf536df69d1f6432cc23deb7acccff86b183f3923a6fd56a1153" +checksum = "fae8ee26b0371a29a77d2b2d6b3ae13aa81def6f9bf1b1b92a32d279a5e709b7" [[package]] name = "rustc-demangle" -version = "0.1.27" +version = "0.1.28" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b50b8869d9fc858ce7266cce0194bd74df58b9d0e3f6df3a9fc8eb470d95c09d" +checksum = "b74b56ffa8bb2830709a538c2cbcae9aa062db0d2a42563bfb09bdaae44020eb" [[package]] name = "rustc-hash" @@ -5879,9 +5910,9 @@ checksum = "08d43f7aa6b08d49f382cde6a7982047c3426db949b1424bc4b7ec9ae12c6ce2" [[package]] name = "rustc-hash" -version = "2.1.2" +version = "2.1.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "94300abf3f1ae2e2b8ffb7b58043de3d399c73fa6f4b73826402a5c457614dbe" +checksum = "6b1e7f9a428571be2dc5bc0505c13fb6bf936822b894ec87abf8a08a4e51742d" [[package]] name = "rustc_version" @@ -5898,7 +5929,7 @@ version = "1.1.4" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "b6fe4565b9518b83ef4f91bb47ce29620ca828bd32cb7e408f0062e9930ba190" dependencies = [ - "bitflags 2.11.1", + "bitflags 2.13.1", "errno", "libc", "linux-raw-sys", @@ -5907,9 +5938,9 @@ dependencies = [ [[package]] name = "rustls" -version = "0.23.40" +version = "0.23.43" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ef86cd5876211988985292b91c96a8f2d298df24e75989a43a3c73f2d4d8168b" +checksum = "0283386ce02abc0151e1761d08802dfe86c173b0b494af5cbc086574e453da06" dependencies = [ "once_cell", "ring", @@ -5921,9 +5952,9 @@ dependencies = [ [[package]] name = "rustls-native-certs" -version = "0.8.3" +version = "0.8.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "612460d5f7bea540c490b2b6395d8e34a953e52b491accd6c86c8164c5932a63" +checksum = "dab5152771c58876a2146916e53e35057e1a4dfa2b9df0f0305b07f611fdea4d" dependencies = [ "openssl-probe", "rustls-pki-types", @@ -5933,9 +5964,9 @@ dependencies = [ [[package]] name = "rustls-pki-types" -version = "1.14.1" +version = "1.15.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "30a7197ae7eb376e574fe940d068c30fe0462554a3ddbe4eca7838e049c937a9" +checksum = "2f4925028c7eb5d1fcdaf196971378ed9d2c1c4efc7dc5d011256f76c99c0a96" dependencies = [ "web-time", "zeroize", @@ -5943,9 +5974,9 @@ dependencies = [ [[package]] name = "rustls-webpki" -version = "0.103.13" +version = "0.103.15" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "61c429a8649f110dddef65e2a5ad240f747e85f7758a6bccc7e5777bd33f756e" +checksum = "f3c3cf1d8b1e7d4927e2d154c3fcb02979afb9939629c62cd9048d4f07b60ac2" dependencies = [ "ring", "rustls-pki-types", @@ -5954,9 +5985,9 @@ dependencies = [ [[package]] name = "rustversion" -version = "1.0.22" +version = "1.0.23" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b39cdef0fa800fc44525c84ccb54a029961a8215f9619753635a9c0d2538d46d" +checksum = "cf54715a573b99ac80df0bc206da022bcd442c974952c7b9720069370852e21f" [[package]] name = "ryu" @@ -6005,13 +6036,13 @@ dependencies = [ [[package]] name = "scroll_derive" -version = "0.13.1" +version = "0.13.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ed76efe62313ab6610570951494bdaa81568026e0318eaa55f167de70eeea67d" +checksum = "e1a36a382ed65dbcc0ab47fd5e9a94112417ccd34560a392ef3b7b0f0ec39148" dependencies = [ "proc-macro2", "quote", - "syn 2.0.117", + "syn 3.0.4", ] [[package]] @@ -6020,7 +6051,7 @@ version = "3.7.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "b7f4bc775c73d9a02cde8bf7b2ec4c9d12743edf609006c7facc23998404cd1d" dependencies = [ - "bitflags 2.11.1", + "bitflags 2.13.1", "core-foundation 0.10.1", "core-foundation-sys", "libc", @@ -6060,9 +6091,9 @@ checksum = "1bc711410fbe7399f390ca1c3b60ad0f53f80e95c5eb935e52268a0e2cd49acc" [[package]] name = "serde" -version = "1.0.228" +version = "1.0.229" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9a8e94ea7f378bd32cbbd37198a4a91436180c5bb472411e48b5ec2e2124ae9e" +checksum = "4148590afebada386688f18773da617792bf2ef03ffc1e4cbd2b1d45b023e0ba" dependencies = [ "serde_core", "serde_derive", @@ -6079,51 +6110,33 @@ dependencies = [ "wasm-bindgen", ] -[[package]] -name = "serde_buf" -version = "0.1.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fc948de1bbead18a61be0b33182636603ea0239ca2577b9704fc39eba900e4e5" -dependencies = [ - "serde_core", -] - [[package]] name = "serde_core" -version = "1.0.228" +version = "1.0.229" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "41d385c7d4ca58e59fc732af25c3983b67ac852c1a25000afe1175de458b67ad" +checksum = "67dca2c9c51e58a4791a4b1ed58308b39c64224d349a935ab5039aa360942a48" dependencies = [ "serde_derive", ] [[package]] name = "serde_derive" -version = "1.0.228" +version = "1.0.229" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d540f220d3187173da220f885ab66608367b6574e925011a9353e4badda91d79" +checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348" dependencies = [ "proc-macro2", "quote", - "syn 2.0.117", -] - -[[package]] -name = "serde_fmt" -version = "1.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6e497af288b3b95d067a23a4f749f2861121ffcb2f6d8379310dcda040c345ed" -dependencies = [ - "serde_core", + "syn 3.0.4", ] [[package]] name = "serde_json" -version = "1.0.149" +version = "1.0.151" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "83fc039473c5595ace860d8c4fafa220ff474b3fc6bfdb4293327f1a37e94d86" +checksum = "c841b55ecdae098c80dcae9cf767f6f8a0c2cdb3416bbef72181df4d0fe73f14" dependencies = [ - "indexmap 2.14.0", + "indexmap", "itoa", "memchr", "serde", @@ -6150,18 +6163,18 @@ checksum = "f3faaf9e727533a19351a43cc5a8de957372163c7d35cc48c90b75cdda13c352" dependencies = [ "percent-encoding", "serde", - "thiserror 2.0.18", + "thiserror 2.0.20", ] [[package]] name = "serde_repr" -version = "0.1.20" +version = "0.1.21" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "175ee3e80ae9982737ca543e96133087cbd9a485eecc3bc4de9c1a37b47ea59c" +checksum = "8d3b1629de253c70a0508c3899572da79ca359fdab27c7920ff00406df418906" dependencies = [ "proc-macro2", "quote", - "syn 2.0.117", + "syn 3.0.4", ] [[package]] @@ -6173,7 +6186,7 @@ dependencies = [ "proc-macro2", "quote", "serde", - "syn 2.0.117", + "syn 2.0.119", ] [[package]] @@ -6190,9 +6203,9 @@ dependencies = [ [[package]] name = "sha1" -version = "0.10.6" +version = "0.10.7" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e3bf829a2d51ab4a5ddf1352d8470c140cadc8301b2ae1789db023f01cedd6ba" +checksum = "a978451301f4db1d02937a4ab3ccce137717b81826e79b7d49ffe3244a13c3b8" dependencies = [ "cfg-if", "cpufeatures 0.2.17", @@ -6217,7 +6230,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "446ba717509524cb3f22f17ecc096f10f4822d76ab5c0b9822c5f9c284e825f4" dependencies = [ "cfg-if", - "cpufeatures 0.3.0", + "cpufeatures 0.3.1", "digest 0.11.3", ] @@ -6232,15 +6245,37 @@ dependencies = [ [[package]] name = "shlex" -version = "1.3.0" +version = "2.0.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0fda2ff0d084019ba4d7c6f371c95d8fd75ce3524c3cb8fb653a3023f6323e64" +checksum = "f8fadd59c855ef2080decdef8ff161eb6661b86933c9d82e5ba29dc602a55aba" [[package]] name = "shuttle" -version = "0.9.1" +version = "0.9.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ba93071c1b720be2505f4c8ce2863502cb9a26a3819e268df1932458a755152c" +checksum = "786792d8bc94c770b53a938f0ae68726387a3a0a3762f9cc2c38d5e1bb40f0c0" +dependencies = [ + "bitvec", + "cfg-if", + "const-siphasher", + "corosensei", + "hex", + "owo-colors", + "rand 0.8.8", + "rand_core 0.6.4", + "rand_pcg 0.3.1", + "scoped-tls", + "shuttle-engine", + "shuttle-schedulers", + "shuttle-std", + "tracing", +] + +[[package]] +name = "shuttle-engine" +version = "0.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bb03d6daa3cf319f53bef382ee4431203c770912209703452033d19fb71e813b" dependencies = [ "assoc", "bitvec", @@ -6249,19 +6284,45 @@ dependencies = [ "corosensei", "hex", "owo-colors", - "rand 0.8.6", + "rand 0.8.8", "rand_core 0.6.4", - "rand_pcg", + "rand_pcg 0.3.1", "scoped-tls", "smallvec", "tracing", ] +[[package]] +name = "shuttle-schedulers" +version = "0.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d7673a579e1660404067f5aa1344d6530f15b59f53b0fe98767d7ab117d14067" +dependencies = [ + "rand 0.8.8", + "rand_pcg 0.3.1", + "shuttle-engine", + "smallvec", + "tracing", +] + +[[package]] +name = "shuttle-std" +version = "0.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bf4a3088fd2d19e5ebc24bc3779bc85a68228fafde2d87d138e475b8591dc721" +dependencies = [ + "assoc", + "owo-colors", + "shuttle-engine", + "smallvec", + "tracing", +] + [[package]] name = "simd-adler32" -version = "0.3.9" +version = "0.3.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "703d5c7ef118737c72f1af64ad2f6f8c5e1921f818cdcb97b8fe6fc69bf66214" +checksum = "3a219298ac11a56ea9a6d2120044824d6f01aeb034955e7af7bc16858527deea" [[package]] name = "simdutf8" @@ -6277,9 +6338,9 @@ checksum = "bbbb5d9659141646ae647b42fe094daf6c6192d1620870b449d9557f748b2daa" [[package]] name = "siphasher" -version = "1.0.2" +version = "1.0.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b2aa850e253778c88a04c3d7323b043aeda9d3e30d5971937c1855769763678e" +checksum = "8ee5873ec9cce0195efcb7a4e9507a04cd49aec9c83d0389df45b1ef7ba2e649" [[package]] name = "slab" @@ -6304,7 +6365,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "bb251b407f50028476a600541542b605bb864d35d9ee1de4f6cab45d88475e6d" dependencies = [ "quote", - "syn 2.0.117", + "syn 2.0.119", ] [[package]] @@ -6334,21 +6395,21 @@ checksum = "88414a5ca1f85d82cc34471e975f0f74f6aa54c40f062efa42c0080e7f763f81" [[package]] name = "smallvec" -version = "1.15.1" +version = "1.15.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "67b1b7a3b5fe4f1376887184045fcf45c69e92af734b7aaddc05fb777b6fbd03" +checksum = "8ed6a63f02c8539c91a8685a86f4099661ba3da017932f6ebbea6de3f0fa7c90" [[package]] name = "snap" -version = "1.1.1" +version = "1.1.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1b6b67fb9a61334225b5b790716f609cd58395f895b3fe8b328786812a40bc3b" +checksum = "199905e6153d6405f9728fe44daace35f8f837bbf830bb6e85fbd5828709a886" [[package]] name = "socket2" -version = "0.6.3" +version = "0.6.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3a766e1110788c36f4fa1c2b71b387a7815aa65f88ce0229841826633d93723e" +checksum = "c3d1e2c7f27f8d4cb10542a02c49005dbd6e93095799d6f3be745fae9f8fedd4" dependencies = [ "libc", "windows-sys 0.61.2", @@ -6356,15 +6417,15 @@ dependencies = [ [[package]] name = "spin" -version = "0.9.8" +version = "0.9.9" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6980e8d7511241f8acf4aebddbb1ff938df5eebe98691418c4468d0b72a96a67" +checksum = "3763264f6b73151db08c50ff20d7d8a0b8796e021cdea7ceedad07b80155fa0e" [[package]] name = "spin" -version = "0.10.0" +version = "0.10.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d5fe4ccb98d9c292d56fec89a5e07da7fc4cf0dc11e156b41793132775d3e591" +checksum = "023a211cb3138dbc438680b32560ad89f699977624c9f8dbb95a47d5b4c07dd3" dependencies = [ "lock_api", ] @@ -6388,7 +6449,7 @@ checksum = "a6dd45d8fc1c79299bfbb7190e42ccbbdf6a5f52e4a6ad98d92357ea965bd289" dependencies = [ "proc-macro2", "quote", - "syn 2.0.117", + "syn 2.0.119", ] [[package]] @@ -6399,9 +6460,9 @@ checksum = "6ce2be8dc25455e1f91df71bfa12ad37d7af1092ae736f3a6cd0e37bc7810596" [[package]] name = "stacker" -version = "0.1.24" +version = "0.1.25" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "640c8cdd92b6b12f5bcb1803ca3bbf5ab96e5e6b6b96b9ab77dabe9e880b3190" +checksum = "707f49d46706bacf8a2b00d51dace3f9de527c13eec3778f570c411f89e69967" dependencies = [ "cc", "cfg-if", @@ -6412,9 +6473,9 @@ dependencies = [ [[package]] name = "str_stack" -version = "0.1.0" +version = "0.1.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9091b6114800a5f2141aee1d1b9d6ca3592ac062dc5decb3764ec5895a47b4eb" +checksum = "7f446288b699d66d0fd2e30d1cfe7869194312524b3b9252594868ed26ef056a" [[package]] name = "strsim" @@ -6424,9 +6485,9 @@ checksum = "7da8b5736845d9f2fcb837ea5d9e2628564b3b043a70948a3f0b778838c5fb4f" [[package]] name = "subsecond" -version = "0.7.6" +version = "0.7.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "feae81a4a7ca6d0bcf70c385a43b7dbacbff527f0805cb0a4043ce2c2c559a2c" +checksum = "d350d5788fa94d560d92269266a50efc77b036bfca6675e420b64df7b3211f37" dependencies = [ "js-sys", "libc", @@ -6435,7 +6496,7 @@ dependencies = [ "memmap2", "serde", "subsecond-types", - "thiserror 2.0.18", + "thiserror 2.0.20", "wasm-bindgen", "wasm-bindgen-futures", "web-sys", @@ -6443,9 +6504,9 @@ dependencies = [ [[package]] name = "subsecond-types" -version = "0.7.6" +version = "0.7.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "85256ee192cbdf00473e48e6133863b125dd4f772fddfbc97287ec7a61458c25" +checksum = "dcf32d66269b5fbb8558334e8d22b657b2f7af0dd2ef56c231541099a862e464" dependencies = [ "serde", ] @@ -6456,84 +6517,6 @@ version = "2.6.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "13c2bddecc57b384dee18652358fb23172facb8a2c51ccc10d74c157bdea3292" -[[package]] -name = "sval" -version = "2.18.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2eb9318255ebd817902d7e279d8f8e39b35b1b9954decd5eb9ea0e30e5fd2b6a" - -[[package]] -name = "sval_buffer" -version = "2.18.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "12571299185e653fdb0fbfe36cd7f6529d39d4e747a60b15a3f34574b7b97c61" -dependencies = [ - "sval", - "sval_ref", -] - -[[package]] -name = "sval_dynamic" -version = "2.18.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "39526f24e997706c0de7f03fb7371f7f5638b66a504ded508e20ad173d0a3677" -dependencies = [ - "sval", -] - -[[package]] -name = "sval_fmt" -version = "2.18.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "933dd3bb26965d682280fcc49400ac2a05036f4ee1e6dbd61bf8402d5a5c3a54" -dependencies = [ - "itoa", - "ryu", - "sval", -] - -[[package]] -name = "sval_json" -version = "2.18.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a0cda08f6d5c9948024a6551077557b1fdcc3880ff2f20ae839667d2ec2d87ed" -dependencies = [ - "itoa", - "ryu", - "sval", -] - -[[package]] -name = "sval_nested" -version = "2.18.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "88d49d5e6c1f9fd0e53515819b03a97ca4eb1bff5c8ee097c43391c09ecfb19f" -dependencies = [ - "sval", - "sval_buffer", - "sval_ref", -] - -[[package]] -name = "sval_ref" -version = "2.18.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "14f876c5a78405375b4e19cbb9554407513b59c93dea12dc6a4af4e1d30899ca" -dependencies = [ - "sval", -] - -[[package]] -name = "sval_serde" -version = "2.18.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5f9ccd3b7f7200239a655e517dd3fd48d960b9111ad24bd6a5e055bef17607c7" -dependencies = [ - "serde_core", - "sval", - "sval_nested", -] - [[package]] name = "symbolic-common" version = "12.18.3" @@ -6559,9 +6542,9 @@ dependencies = [ [[package]] name = "syn" -version = "2.0.117" +version = "2.0.119" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e665b8803e7b1d2a727f4023456bbbbe74da67099c585258af0ad9c5013b9b99" +checksum = "872831b642d1a07999a962a351ed35b955ea2cfc8f3862091e2a240a84f17297" dependencies = [ "proc-macro2", "quote", @@ -6596,20 +6579,21 @@ checksum = "728a70f3dbaf5bab7f0c4b1ac8d7ae5ea60a4b5549c8a5914361c99147a709d2" dependencies = [ "proc-macro2", "quote", - "syn 2.0.117", + "syn 2.0.119", ] [[package]] name = "sysinfo" -version = "0.38.4" +version = "0.39.6" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "92ab6a2f8bfe508deb3c6406578252e491d299cbbf3bc0529ecc3313aee4a52f" +checksum = "d2071df9448915b71c4fe6d25deaf1c22f12bd234f01540b77312bb8e41361e6" dependencies = [ "libc", "memchr", "ntapi", "objc2-core-foundation", "objc2-io-kit", + "objc2-open-directory", "windows", ] @@ -6619,7 +6603,7 @@ version = "0.7.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "a13f3d0daba03132c0aa9767f98351b3488edc2c100cda2d2ec2b04f3d8d3c8b" dependencies = [ - "bitflags 2.11.1", + "bitflags 2.13.1", "core-foundation 0.9.4", "system-configuration-sys", ] @@ -6636,9 +6620,9 @@ dependencies = [ [[package]] name = "t4" -version = "0.1.7" +version = "0.1.9" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d3105026c5fb7e1dcf8864c2193387439aa650b690e1a5513ab27ae7de7999a9" +checksum = "5d116b5de6db7e42465e0b9807038609248e536e84bc76167b65622dac1dc635" dependencies = [ "cfg-if", "crossbeam-channel", @@ -6652,9 +6636,9 @@ dependencies = [ [[package]] name = "t4-verified" -version = "0.1.7" +version = "0.1.9" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "494c2f2c4cfb21784ef9b380c378fbfdbbe2f19f1ad606e6729816ea2a0f4297" +checksum = "9610c43f95b308dc0f5d717b84698783869980c9ca23ce19f78ec7b72a5dbf2b" dependencies = [ "vstd", ] @@ -6672,7 +6656,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "32497e9a4c7b38532efcdebeef879707aa9f794296a4f0244f6f69e9bc8574bd" dependencies = [ "fastrand", - "getrandom 0.4.2", + "getrandom 0.4.3", "once_cell", "rustix", "windows-sys 0.61.2", @@ -6699,11 +6683,11 @@ dependencies = [ [[package]] name = "thiserror" -version = "2.0.18" +version = "2.0.20" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4288b5bcbc7920c07a1149a35cf9590a2aa808e0bc1eafaade0b80947865fbc4" +checksum = "ec86235f5fcc2a73650310756d2ac5b138a5780bbbdfae3eeccec992c435ba4f" dependencies = [ - "thiserror-impl 2.0.18", + "thiserror-impl 2.0.20", ] [[package]] @@ -6714,18 +6698,18 @@ checksum = "4fee6c4efc90059e10f81e6d42c60a18f76588c3d74cb83a0b242a2b6c7504c1" dependencies = [ "proc-macro2", "quote", - "syn 2.0.117", + "syn 2.0.119", ] [[package]] name = "thiserror-impl" -version = "2.0.18" +version = "2.0.20" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ebc4ee7f67670e9b64d05fa4253e753e016c6c95ff35b89b7941d6b856dec1d5" +checksum = "bc04cd3e1236dd4a98afca4569f2deb3f120e5422a4023be2cb683f8486292af" dependencies = [ "proc-macro2", "quote", - "syn 2.0.117", + "syn 3.0.4", ] [[package]] @@ -6740,21 +6724,20 @@ dependencies = [ [[package]] name = "thread_local" -version = "1.1.9" +version = "1.1.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f60246a4944f24f6e018aa17cdeffb7818b76356965d03b07d6a9886e8962185" +checksum = "1ad99c4c6d32803332c548b1af0540b357b3f5fc0be8f6c6bfe8b2e6ae784070" dependencies = [ "cfg-if", ] [[package]] name = "time" -version = "0.3.47" +version = "0.3.55" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "743bd48c283afc0388f9b8827b976905fb217ad9e647fae3a379a9283c4def2c" +checksum = "cdb87b95ec50ddfa440816d227a17b2ccbdda963a316a727fda0fc4334f7d134" dependencies = [ "deranged", - "itoa", "num-conv", "powerfmt", "serde_core", @@ -6764,15 +6747,15 @@ dependencies = [ [[package]] name = "time-core" -version = "0.1.8" +version = "0.1.9" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7694e1cfe791f8d31026952abf09c69ca6f6fa4e1a1229e18988f06a04a12dca" +checksum = "9e1c906769ad99c88eaa54e728060edef082f8e358ff32030cb7c7d315e81109" [[package]] name = "time-macros" -version = "0.2.27" +version = "0.2.32" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2e70e4c5a0e0a8a4823ad65dfe1a6930e4f4d756dcd9dd7939022b5e8c501215" +checksum = "7e689342a48d2ea927c87ea50cabf8594854bf940e9310208848d680d668ed85" dependencies = [ "num-conv", "time-core", @@ -6789,9 +6772,9 @@ dependencies = [ [[package]] name = "tinystr" -version = "0.8.3" +version = "0.8.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c8323304221c2a851516f22236c5722a72eaa19749016521d6dff0824447d96d" +checksum = "b1e27c91459209c2986af3dcf603a5a74a4368754ce37414f59acc971167f643" dependencies = [ "displaydoc", "zerovec", @@ -6799,9 +6782,9 @@ dependencies = [ [[package]] name = "tinyvec" -version = "1.11.0" +version = "1.12.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3e61e67053d25a4e82c844e8424039d9745781b3fc4f32b8d55ed50f5f667ef3" +checksum = "bb4ebadaa0af04fab11ae01eb5f9fdb5f9c5b875506e210e71c07873528baa7f" dependencies = [ "tinyvec_macros", ] @@ -6814,9 +6797,9 @@ checksum = "1f3ccbac311fea05f86f61904b462b55fb3df8837a366dfc601a0161d0532f20" [[package]] name = "tokio" -version = "1.52.3" +version = "1.53.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8fc7f01b389ac15039e4dc9531aa973a135d7a4135281b12d7c1bc79fd57fffe" +checksum = "202caea871b69668250d242070849eb495be178ed697a3e98aebce5bc81a0bed" dependencies = [ "bytes", "libc", @@ -6829,13 +6812,13 @@ dependencies = [ [[package]] name = "tokio-macros" -version = "2.7.0" +version = "2.7.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "385a6cb71ab9ab790c5fe8d67f1645e6c450a7ce006a33de03daa956cf70a496" +checksum = "78773a2a397f451582ce068015985c33193cf6dea8b74d2a639fe457b2f07b0e" dependencies = [ "proc-macro2", "quote", - "syn 2.0.117", + "syn 3.0.4", ] [[package]] @@ -6850,9 +6833,9 @@ dependencies = [ [[package]] name = "tokio-stream" -version = "0.1.18" +version = "0.1.19" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "32da49809aab5c3bc678af03902d4ccddea2a87d028d86392a4b1560c6906c70" +checksum = "a3d06f0b082ba57c26b79407372e57cf2a1e28124f78e9479fe80322cf53420b" dependencies = [ "futures-core", "pin-project-lite", @@ -6897,15 +6880,16 @@ dependencies = [ [[package]] name = "tokio-util" -version = "0.7.18" +version = "0.7.19" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9ae9cec805b01e8fc3fd2fe289f89149a9b66dd16786abd8b19cfa7b48cb0098" +checksum = "494815d09bf52b5548659851081238f0ca39ff638363907596da739561c62c52" dependencies = [ "bytes", "futures-core", "futures-io", "futures-sink", "futures-util", + "libc", "pin-project-lite", "tokio", ] @@ -6921,23 +6905,23 @@ dependencies = [ [[package]] name = "toml_edit" -version = "0.25.11+spec-1.1.0" +version = "0.25.13+spec-1.1.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0b59c4d22ed448339746c59b905d24568fcbb3ab65a500494f7b8c3e97739f2b" +checksum = "6975367e4d2ef766d86af01ffad14b622fecc8d4357a998fbc4deb6e9bacaf9b" dependencies = [ - "indexmap 2.14.0", + "indexmap", "toml_datetime", "toml_parser", - "winnow 1.0.2", + "winnow 1.0.4", ] [[package]] name = "toml_parser" -version = "1.1.2+spec-1.1.0" +version = "1.1.3+spec-1.1.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a2abe9b86193656635d2411dc43050282ca48aa31c2451210f4202550afb7526" +checksum = "1d38ac1cf9b95face32296c0a3ede1fdc270627c9d9c02a7274dd6d960dc4d56" dependencies = [ - "winnow 1.0.2", + "winnow 1.0.4", ] [[package]] @@ -6971,15 +6955,26 @@ dependencies = [ [[package]] name = "tonic-prost" -version = "0.14.5" +version = "0.14.6" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a55376a0bbaa4975a3f10d009ad763d8f4108f067c7c2e74f3001fb49778d309" +checksum = "50849f68853be452acf590cde0b146665b8d507b3b8af17261df47e02c209ea0" dependencies = [ "bytes", "prost", "tonic", ] +[[package]] +name = "tonic-types" +version = "0.14.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "73ab1b02061f83d519bba3caa167f88f261ef05720ab8ebc954ade70de3348e8" +dependencies = [ + "prost", + "prost-types", + "tonic", +] + [[package]] name = "tower" version = "0.5.3" @@ -6988,7 +6983,7 @@ checksum = "ebe5ef63511595f1344e2d5cfa636d973292adc0eec1f0ad45fae9f0851ab1d4" dependencies = [ "futures-core", "futures-util", - "indexmap 2.14.0", + "indexmap", "pin-project-lite", "slab", "sync_wrapper", @@ -7005,7 +7000,7 @@ version = "0.6.11" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "4cfcf7e2740e6fc6d4d688b4ef00650406bb94adf4731e43c096c3a19fe40840" dependencies = [ - "bitflags 2.11.1", + "bitflags 2.13.1", "bytes", "futures-core", "futures-util", @@ -7026,6 +7021,21 @@ dependencies = [ "url", ] +[[package]] +name = "tower-http" +version = "0.7.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "08a05a66a4fdd61cbbe0a1d755ffe0ca6aba159dd4820936a0ff8a8278245b9c" +dependencies = [ + "bitflags 2.13.1", + "bytes", + "http", + "percent-encoding", + "pin-project-lite", + "tower-layer", + "tower-service", +] + [[package]] name = "tower-layer" version = "0.3.3" @@ -7058,7 +7068,7 @@ checksum = "7490cfa5ec963746568740651ac6781f701c9c5ea257c58e057f3ba8cf69e8da" dependencies = [ "proc-macro2", "quote", - "syn 2.0.117", + "syn 2.0.119", ] [[package]] @@ -7138,9 +7148,9 @@ dependencies = [ "http", "httparse", "log", - "rand 0.9.4", + "rand 0.9.5", "sha1", - "thiserror 2.0.18", + "thiserror 2.0.20", "utf-8", ] @@ -7155,9 +7165,9 @@ dependencies = [ "http", "httparse", "log", - "rand 0.9.4", + "rand 0.9.5", "sha1", - "thiserror 2.0.18", + "thiserror 2.0.20", "utf-8", ] @@ -7172,28 +7182,22 @@ dependencies = [ "http", "httparse", "log", - "rand 0.9.4", + "rand 0.9.5", "sha1", - "thiserror 2.0.18", + "thiserror 2.0.20", ] [[package]] name = "twox-hash" -version = "2.1.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9ea3136b675547379c4bd395ca6b938e5ad3c3d20fad76e7fe85f9e0d011419c" - -[[package]] -name = "typeid" -version = "1.0.3" +version = "2.1.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bc7d623258602320d5c55d1bc22793b57daff0ec7efc270ea7d55ce1d5f5471c" +checksum = "5283634e518fe9e82c7b20520bb4bc209009fd16c82077c802f8111ecbb0117a" [[package]] name = "typenum" -version = "1.20.0" +version = "1.20.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "40ce102ab67701b8526c123c1bab5cbe42d7040ccfd0f64af1a385808d2f43de" +checksum = "b6f5e870be6c3b371b77fe0ee0bafb859fa4964b4404c27de1d380043c4dda20" [[package]] name = "ucd-trie" @@ -7215,9 +7219,9 @@ checksum = "e6e4313cd5fcd3dad5cafa179702e2b244f760991f45397d14d4ebf38247da75" [[package]] name = "unicode-segmentation" -version = "1.13.2" +version = "1.13.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9629274872b2bfaf8d66f5f15725007f635594914870f65218920345aa11aa8c" +checksum = "c6f5d3c3b1bf09027a88a6bc961fc00497d651009560b5463668dc81b0fa87a8" [[package]] name = "unicode-width" @@ -7274,7 +7278,7 @@ dependencies = [ "proc-macro2", "quote", "serde_tokenstream", - "syn 2.0.117", + "syn 2.0.119", "usdt-impl", ] @@ -7292,8 +7296,8 @@ dependencies = [ "quote", "serde", "serde_json", - "syn 2.0.117", - "thiserror 2.0.18", + "syn 2.0.119", + "thiserror 2.0.20", "thread-id", ] @@ -7307,7 +7311,7 @@ dependencies = [ "proc-macro2", "quote", "serde_tokenstream", - "syn 2.0.117", + "syn 2.0.119", "usdt-impl", ] @@ -7331,11 +7335,11 @@ checksum = "06abde3611657adf66d383f00b093d7faecc7fa57071cce2578660c9f1010821" [[package]] name = "uuid" -version = "1.23.3" +version = "1.26.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "144d6b123cef80b301b8f72a9e2ca4370ddec21950d0a103dd22c437006d2db7" +checksum = "b5772d71c9be8a8a6ac2117d949c5b224c1b72241bb611d9a3012edcf8af7812" dependencies = [ - "getrandom 0.4.2", + "getrandom 0.4.3", "js-sys", "wasm-bindgen", ] @@ -7346,43 +7350,6 @@ version = "0.1.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "ba73ea9cf16a25df0c8caa16c51acb937d5712a8429db78a3ee29d5dcacd3a65" -[[package]] -name = "value-bag" -version = "1.12.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7ba6f5989077681266825251a52748b8c1d8a4ad098cc37e440103d0ea717fc0" -dependencies = [ - "value-bag-serde1", - "value-bag-sval2", -] - -[[package]] -name = "value-bag-serde1" -version = "1.12.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "16530907bfe2999a1773ca5900a65101e092c70f642f25cc23ca0c43573262c5" -dependencies = [ - "erased-serde", - "serde_buf", - "serde_core", - "serde_fmt", -] - -[[package]] -name = "value-bag-sval2" -version = "1.12.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d00ae130edd690eaa877e4f40605d534790d1cf1d651e7685bd6a144521b251f" -dependencies = [ - "sval", - "sval_buffer", - "sval_dynamic", - "sval_fmt", - "sval_json", - "sval_ref", - "sval_serde", -] - [[package]] name = "version_check" version = "0.9.5" @@ -7391,19 +7358,20 @@ checksum = "0b928f33d975fc6ad9f86c8f283853ad26bdd5b10b7f1542aa2fa15e2289105a" [[package]] name = "verus_builtin" -version = "0.0.0-2026-05-06-1803" +version = "0.0.0-2026-08-30-0159" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "650bcc71ef90cbc79bc2a2494c54dd159edad63a72677be791954edc993883fe" +checksum = "300d269a2e06dbe54cb53084464065912ad7accbb52ae8d6c8aa7a6c521974c5" [[package]] name = "verus_builtin_macros" -version = "0.0.0-2026-05-10-0145" +version = "0.0.0-2026-08-30-0159" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7ef946d78c84f284991d59035ca252f29cd8071526003e794be80f0beee18f2d" +checksum = "cc040bfded82e79a708c1819c52b386ee23c0c823239334f39ced152bbdf3624" dependencies = [ + "convert_case 0.4.0", "proc-macro2", "quote", - "syn 2.0.117", + "syn 2.0.119", "synstructure", "verus_prettyplease", "verus_syn", @@ -7411,9 +7379,9 @@ dependencies = [ [[package]] name = "verus_prettyplease" -version = "0.0.0-2026-05-10-0145" +version = "0.0.0-2026-08-09-0044" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6a227e7eaa03f51ac4d659641192bc1b9fe4eda0a509a6734752cf72f5d7daaf" +checksum = "51fc115de5fb3806bc362060bb683024660ed2bc13c1183d1f01d464e4537139" dependencies = [ "proc-macro2", "verus_syn", @@ -7421,11 +7389,11 @@ dependencies = [ [[package]] name = "verus_state_machines_macros" -version = "0.0.0-2026-05-10-0145" +version = "0.0.0-2026-08-02-0125" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3ff46046bffd2d55757503feef0b72a74a60ead2b58aa211861267485b955136" +checksum = "93f77ff8d121edb1bf2651335769a80a2f611da8573c3b498434a66b3162805f" dependencies = [ - "indexmap 1.9.3", + "indexmap", "proc-macro2", "quote", "verus_syn", @@ -7433,9 +7401,9 @@ dependencies = [ [[package]] name = "verus_syn" -version = "0.0.0-2026-05-10-0145" +version = "0.0.0-2026-08-02-0125" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a82174e474e1f06418dd304f714785ae1c3e4d2fbff415a0d0c05a02cd995819" +checksum = "f17237ea6d267e457d53ce36f55b315fc9edd26eb88c765dd95e035c0b415869" dependencies = [ "proc-macro2", "quote", @@ -7444,9 +7412,9 @@ dependencies = [ [[package]] name = "vstd" -version = "0.0.0-2026-05-10-0145" +version = "0.0.0-2026-08-30-0159" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6958222eaf7bbe565c93890676bb1f96c4f5e25d90a81bb9993fdc19fc0d3f72" +checksum = "7169790c92857fa9441c8ba7e0c469f4fc62104caf1dfc632c318577a1a08d7a" dependencies = [ "verus_builtin", "verus_builtin_macros", @@ -7491,7 +7459,7 @@ checksum = "59195a1db0e95b920366d949ba5e0d3fc0e70b67c09be15ce5abb790106b0571" dependencies = [ "proc-macro2", "quote", - "syn 2.0.117", + "syn 2.0.119", ] [[package]] @@ -7502,27 +7470,18 @@ checksum = "ccf3ec651a847eb01de73ccad15eb7d99f80485de043efb2f370cd654f4ea44b" [[package]] name = "wasip2" -version = "1.0.3+wasi-0.2.9" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "20064672db26d7cdc89c7798c48a0fdfac8213434a1186e5ef29fd560ae223d6" -dependencies = [ - "wit-bindgen 0.57.1", -] - -[[package]] -name = "wasip3" -version = "0.4.0+wasi-0.3.0-rc-2026-01-06" +version = "1.0.4+wasi-0.2.12" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5428f8bf88ea5ddc08faddef2ac4a67e390b88186c703ce6dbd955e1c145aca5" +checksum = "b67efb37e106e55ce722a510d6b5f9c17f083e5fc79afc2badeb12cc313d9487" dependencies = [ - "wit-bindgen 0.51.0", + "wit-bindgen", ] [[package]] name = "wasm-bindgen" -version = "0.2.118" +version = "0.2.126" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0bf938a0bacb0469e83c1e148908bd7d5a6010354cf4fb73279b7447422e3a89" +checksum = "4b067c0c11094aef6b7a801c1e34a26affafdf3d051dba08456b868789aaf9a4" dependencies = [ "cfg-if", "once_cell", @@ -7533,9 +7492,9 @@ dependencies = [ [[package]] name = "wasm-bindgen-futures" -version = "0.4.68" +version = "0.4.76" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f371d383f2fb139252e0bfac3b81b265689bf45b6874af544ffa4c975ac1ebf8" +checksum = "c62df1340f32221cb9c54d6a27b030e3dba64361d4a95bed55f9aacb44da291d" dependencies = [ "js-sys", "wasm-bindgen", @@ -7543,9 +7502,9 @@ dependencies = [ [[package]] name = "wasm-bindgen-macro" -version = "0.2.118" +version = "0.2.126" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "eeff24f84126c0ec2db7a449f0c2ec963c6a49efe0698c4242929da037ca28ed" +checksum = "167ce5e579f6bcf889c4f7175a8a5a585de84e8ff93976ce393efa5f2837aab1" dependencies = [ "quote", "wasm-bindgen-macro-support", @@ -7553,48 +7512,26 @@ dependencies = [ [[package]] name = "wasm-bindgen-macro-support" -version = "0.2.118" +version = "0.2.126" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9d08065faf983b2b80a79fd87d8254c409281cf7de75fc4b773019824196c904" +checksum = "f3997c7839262f4ef12cf90b818d6340c18e80f263f1a94bf157d0ec4420380e" dependencies = [ "bumpalo", "proc-macro2", "quote", - "syn 2.0.117", + "syn 2.0.119", "wasm-bindgen-shared", ] [[package]] name = "wasm-bindgen-shared" -version = "0.2.118" +version = "0.2.126" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5fd04d9e306f1907bd13c6361b5c6bfc7b3b3c095ed3f8a9246390f8dbdee129" +checksum = "dc1b4cb0cc549fcf58d7dfc081778139b3d283a081644e833e84682ad71cea24" dependencies = [ "unicode-ident", ] -[[package]] -name = "wasm-encoder" -version = "0.244.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "990065f2fe63003fe337b932cfb5e3b80e0b4d0f5ff650e6985b1048f62c8319" -dependencies = [ - "leb128fmt", - "wasmparser", -] - -[[package]] -name = "wasm-metadata" -version = "0.244.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bb0e353e6a2fbdc176932bbaab493762eb1255a7900fe0fea1a2f96c296cc909" -dependencies = [ - "anyhow", - "indexmap 2.14.0", - "wasm-encoder", - "wasmparser", -] - [[package]] name = "wasm-streams" version = "0.4.2" @@ -7608,23 +7545,11 @@ dependencies = [ "web-sys", ] -[[package]] -name = "wasmparser" -version = "0.244.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "47b807c72e1bac69382b3a6fb3dbe8ea4c0ed87ff5629b8685ae6b9a611028fe" -dependencies = [ - "bitflags 2.11.1", - "hashbrown 0.15.5", - "indexmap 2.14.0", - "semver", -] - [[package]] name = "web-sys" -version = "0.3.95" +version = "0.3.103" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4f2dfbb17949fa2088e5d39408c48368947b86f7834484e87b73de55bc14d97d" +checksum = "8622dcb61c0bcc9fffa6938bed81210af2da9a7e4a1a834b2e37a59b6dfb6141" dependencies = [ "js-sys", "wasm-bindgen", @@ -7642,9 +7567,9 @@ dependencies = [ [[package]] name = "webpki-roots" -version = "1.0.7" +version = "1.0.9" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "52f5ee44c96cf55f1b349600768e3ece3a8f26010c05265ab73f945bb1a2eb9d" +checksum = "7dcd9d09a39985f5344844e66b0c530a33843579125f23e21e9f0f220850f22a" dependencies = [ "rustls-pki-types", ] @@ -7733,7 +7658,7 @@ checksum = "053e2e040ab57b9dc951b72c264860db7eb3b0200ba345b4e4c3b14f67855ddf" dependencies = [ "proc-macro2", "quote", - "syn 2.0.117", + "syn 2.0.119", ] [[package]] @@ -7744,7 +7669,7 @@ checksum = "3f316c4a2570ba26bbec722032c4099d8c8bc095efccdc15688708623367e358" dependencies = [ "proc-macro2", "quote", - "syn 2.0.117", + "syn 2.0.119", ] [[package]] @@ -7969,112 +7894,24 @@ dependencies = [ [[package]] name = "winnow" -version = "1.0.2" +version = "1.0.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2ee1708bef14716a11bae175f579062d4554d95be2c6829f518df847b7b3fdd0" +checksum = "23b97319f7b8343df12cc98938e5c3eb436064524c8d2b4e30a1d3a36eecdf81" dependencies = [ "memchr", ] -[[package]] -name = "wit-bindgen" -version = "0.51.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d7249219f66ced02969388cf2bb044a09756a083d0fab1e566056b04d9fbcaa5" -dependencies = [ - "wit-bindgen-rust-macro", -] - [[package]] name = "wit-bindgen" version = "0.57.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "1ebf944e87a7c253233ad6766e082e3cd714b5d03812acc24c318f549614536e" -[[package]] -name = "wit-bindgen-core" -version = "0.51.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ea61de684c3ea68cb082b7a88508a8b27fcc8b797d738bfc99a82facf1d752dc" -dependencies = [ - "anyhow", - "heck", - "wit-parser", -] - -[[package]] -name = "wit-bindgen-rust" -version = "0.51.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b7c566e0f4b284dd6561c786d9cb0142da491f46a9fbed79ea69cdad5db17f21" -dependencies = [ - "anyhow", - "heck", - "indexmap 2.14.0", - "prettyplease", - "syn 2.0.117", - "wasm-metadata", - "wit-bindgen-core", - "wit-component", -] - -[[package]] -name = "wit-bindgen-rust-macro" -version = "0.51.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0c0f9bfd77e6a48eccf51359e3ae77140a7f50b1e2ebfe62422d8afdaffab17a" -dependencies = [ - "anyhow", - "prettyplease", - "proc-macro2", - "quote", - "syn 2.0.117", - "wit-bindgen-core", - "wit-bindgen-rust", -] - -[[package]] -name = "wit-component" -version = "0.244.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9d66ea20e9553b30172b5e831994e35fbde2d165325bec84fc43dbf6f4eb9cb2" -dependencies = [ - "anyhow", - "bitflags 2.11.1", - "indexmap 2.14.0", - "log", - "serde", - "serde_derive", - "serde_json", - "wasm-encoder", - "wasm-metadata", - "wasmparser", - "wit-parser", -] - -[[package]] -name = "wit-parser" -version = "0.244.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ecc8ac4bc1dc3381b7f59c34f00b67e18f910c2c0f50015669dde7def656a736" -dependencies = [ - "anyhow", - "id-arena", - "indexmap 2.14.0", - "log", - "semver", - "serde", - "serde_derive", - "serde_json", - "unicode-xid", - "wasmparser", -] - [[package]] name = "writeable" -version = "0.6.3" +version = "0.6.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1ffae5123b2d3fc086436f8834ae3ab053a283cfac8fe0a0b8eaae044768a4c4" +checksum = "3ad82d2a33cdc9674dc7465672f271e096168fcdbe0f799d9e6db8c5892679dc" [[package]] name = "wyz" @@ -8087,15 +7924,15 @@ dependencies = [ [[package]] name = "xxhash-rust" -version = "0.8.15" +version = "0.8.18" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fdd20c5420375476fbd4394763288da7eb0cc0b8c11deed431a91562af7335d3" +checksum = "aee1b19627c7c60102ab80d3a9cbe18de90bfe03bfa6c3715447681f0e8c8af6" [[package]] name = "yoke" -version = "0.8.2" +version = "0.8.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "abe8c5fda708d9ca3df187cae8bfb9ceda00dd96231bed36e445a1a48e66f9ca" +checksum = "709fe23a0424b6a435d82152b1bd3fdfb0833487d5fa90d05d42762a9891fef5" dependencies = [ "stable_deref_trait", "yoke-derive", @@ -8110,35 +7947,35 @@ checksum = "de844c262c8848816172cef550288e7dc6c7b7814b4ee56b3e1553f275f1858e" dependencies = [ "proc-macro2", "quote", - "syn 2.0.117", + "syn 2.0.119", "synstructure", ] [[package]] name = "zerocopy" -version = "0.8.48" +version = "0.8.56" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "eed437bf9d6692032087e337407a86f04cd8d6a16a37199ed57949d415bd68e9" +checksum = "556764e583adb45a9f8d413c2a147fa7e8d821e48e12b14fd560b607998b75eb" dependencies = [ "zerocopy-derive", ] [[package]] name = "zerocopy-derive" -version = "0.8.48" +version = "0.8.56" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "70e3cd084b1788766f53af483dd21f93881ff30d7320490ec3ef7526d203bad4" +checksum = "f2ab42fc20575779bd240faa45f94a74256f755c0fa9e89f0ede20d91d0cdfc1" dependencies = [ "proc-macro2", "quote", - "syn 2.0.117", + "syn 2.0.119", ] [[package]] name = "zerofrom" -version = "0.1.7" +version = "0.1.8" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "69faa1f2a1ea75661980b013019ed6687ed0e83d069bc1114e2cc74c6c04c4df" +checksum = "0ec05a11813ea801ff6d75110ad09cd0824ddba17dfe17128ea0d5f68e6c5272" dependencies = [ "zerofrom-derive", ] @@ -8151,21 +7988,21 @@ checksum = "11532158c46691caf0f2593ea8358fed6bbf68a0315e80aae9bd41fbade684a1" dependencies = [ "proc-macro2", "quote", - "syn 2.0.117", + "syn 2.0.119", "synstructure", ] [[package]] name = "zeroize" -version = "1.8.2" +version = "1.9.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b97154e67e32c85465826e8bcc1c59429aaaf107c1e4a9e53c8d8ccd5eff88d0" +checksum = "e13c156562582aa81c60cb29407084cdb54c4164760106ab78e6c5b0858cf64e" [[package]] name = "zerotrie" -version = "0.2.4" +version = "0.2.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0f9152d31db0792fa83f70fb2f83148effb5c1f5b8c7686c3459e361d9bc20bf" +checksum = "4ea269c3bd32f0a32c321907a2ae912ba6f4649bb0fc764a15627e99a7095a3f" dependencies = [ "displaydoc", "yoke", @@ -8174,9 +8011,9 @@ dependencies = [ [[package]] name = "zerovec" -version = "0.11.6" +version = "0.11.8" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "90f911cbc359ab6af17377d242225f4d75119aec87ea711a880987b18cd7b239" +checksum = "bb0464e17806c1d976d5cba29399c7f08e516e279e2ba493f63123b5fca67dd8" dependencies = [ "yoke", "zerofrom", @@ -8185,26 +8022,26 @@ dependencies = [ [[package]] name = "zerovec-derive" -version = "0.11.3" +version = "0.11.6" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "625dc425cab0dca6dc3c3319506e6593dcb08a9f387ea3b284dbd52a92c40555" +checksum = "34df6fc39dbd26ddc9c10e6a2984476e13acce22e64e4487636ef494369225da" dependencies = [ "proc-macro2", "quote", - "syn 2.0.117", + "syn 3.0.4", ] [[package]] name = "zlib-rs" -version = "0.6.3" +version = "0.6.7" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3be3d40e40a133f9c916ee3f9f4fa2d9d63435b5fbe1bfc6d9dae0aa0ada1513" +checksum = "34b31d188d9d685a4f9c7b46d6e36631b07058d2cfe190267adce54dc230bf12" [[package]] name = "zmij" -version = "1.0.21" +version = "1.0.23" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b8848ee67ecc8aedbaf3e4122217aff892639231befc6a1b58d29fff4c2cabaa" +checksum = "29666d0abbfad1e3dc4dcf6144730dd3a3ab225bbbdac83319345b1b44ccfc1b" [[package]] name = "zstd" diff --git a/Cargo.toml b/Cargo.toml index ade6052ab..4345181d6 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -1,6 +1,10 @@ [workspace.package] version = "0.1.13" edition = "2024" +# sysinfo 0.39 is the current floor; keep this in step with the oldest +# toolchain the dependency tree accepts so a stale rustc fails with +# cargo's own message rather than a deep type error. +rust-version = "1.95" repository = "https://github.com/XiangpengHao/liquid-cache" authors = ["XiangpengHao "] license = "Apache-2.0 OR MIT" @@ -30,25 +34,26 @@ liquid-cache-datafusion = { path = "src/datafusion", version = "0.1.13" } liquid-cache-common = { path = "src/common", version = "0.1.13" } liquid-cache = { path = "src/core", version = "0.1.13" } liquid-cache-datafusion-local = { path = "src/datafusion-local", version = "0.1.13" } -arrow = { version = "59.2", default-features = false, features = [ +arrow = { version = "59.2.0", default-features = false, features = [ "prettyprint", "ipc", ] } -arrow-flight = { version = "59.2", features = ["flight-sql-experimental"] } -arrow-schema = { version = "59.2", features = ["serde"] } -parquet = { version = "59.2", features = [ +arrow-flight = { version = "59.2.0", features = ["flight-sql-experimental"] } +arrow-schema = { version = "59.2.0", features = ["serde"] } +parquet = { version = "59.2.0", features = [ "async", "experimental", "variant_experimental", ] } -parquet-variant-json = { version = "59.2" } -parquet-variant-compute = { version = "59.2" } -datafusion = { version = "55" } -datafusion-common = { version = "55" } -datafusion-expr-common = { version = "55" } -datafusion-physical-expr = { version = "55" } -datafusion-physical-expr-common = { version = "55" } -datafusion-proto = { version = "55" } +parquet-variant-json = { version = "59.2.0" } +parquet-variant-compute = { version = "59.2.0" } +datafusion = { version = "55.0.0" } +datafusion-datasource = { version = "55.0.0" } +datafusion-common = { version = "55.0.0" } +datafusion-expr-common = { version = "55.0.0" } +datafusion-physical-expr = { version = "55.0.0" } +datafusion-physical-expr-common = { version = "55.0.0" } +datafusion-proto = { version = "55.0.0" } async-trait = "0.1.89" futures = { version = "0.3.32", default-features = false, features = ["std"] } tokio = { version = "1.52.3", features = ["rt-multi-thread"] } @@ -67,7 +72,7 @@ fastrace = "0.7" fastrace-tonic = "0.2" congee = "0.4.1" insta = "1.47.2" -t4 = "0.1.7" +t4 = "0.1.9" [profile.dev.package] insta.opt-level = 3 diff --git a/README.md b/README.md index aa1c86a95..7cb7b64ad 100644 --- a/README.md +++ b/README.md @@ -93,7 +93,7 @@ tokio_test::block_on(async { ### LiquidCache uses DIRECT I/O -On Linux, LiquidCache uses [DIRECT I/O](https://man7.org/linux/man-pages/man2/open.2.html#:~:text=O_DIRECT). This means that it bypasses the OS page cache, this avoids double-caching and bound memory usage. +By default, LiquidCache bypasses the OS page cache using [O_DIRECT](https://man7.org/linux/man-pages/man2/open.2.html#:~:text=O_DIRECT) on Linux and `F_NOCACHE` on macOS. This avoids double-caching and bounds memory usage. This also means LiquidCache can *appear slower* than other caches when most data fits in OS page cache, which is common in dev environments but unrealistic in production. diff --git a/benchmark/Cargo.toml b/benchmark/Cargo.toml index 870bdb4c1..6a2a97c03 100644 --- a/benchmark/Cargo.toml +++ b/benchmark/Cargo.toml @@ -2,6 +2,7 @@ name = "liquid-cache-benchmarks" description = "LiquidCache Benchmarks" edition = { workspace = true } +rust-version = { workspace = true } publish = false [dependencies] @@ -21,7 +22,7 @@ url = { workspace = true } mimalloc = "0.1.52" serde_json.workspace = true serde.workspace = true -sysinfo = { version = "0.38.4", default-features = false, features = [ +sysinfo = { version = "0.39.6", default-features = false, features = [ "network", "disk", ] } @@ -30,11 +31,11 @@ parquet = { workspace = true } arrow = { workspace = true } fastrace = { version = "0.7.17" } fastrace-tonic = { workspace = true } -fastrace-opentelemetry = "0.16" -opentelemetry = "0.31.0" -opentelemetry_sdk = "0.31.0" -opentelemetry-otlp = { version = "0.31.1", features = ["trace", "grpc-tonic"] } -logforth = { version = "0.29.1", features = ["append-opentelemetry", "bridge-log"] } +fastrace-opentelemetry = "0.18" +opentelemetry = "0.32.0" +opentelemetry_sdk = "0.32.0" +opentelemetry-otlp = { version = "0.32.0", features = ["trace", "grpc-tonic"] } +logforth = { version = "0.30.1", features = ["starter-log", "filter-rustlog"] } reqwest = { version = "0.13.4", default-features = false, features = ["json"] } uuid = { version = "1.23.3", features = ["v4"] } pprof = { version = "0.15.0", features = ["flamegraph"] } diff --git a/benchmark/src/observability.rs b/benchmark/src/observability.rs index d33d3d455..aff918c84 100644 --- a/benchmark/src/observability.rs +++ b/benchmark/src/observability.rs @@ -7,7 +7,7 @@ use datafusion::datasource::source::DataSource; use datafusion::physical_plan::ExecutionPlan; use fastrace_opentelemetry::OpenTelemetryReporter; use liquid_cache_datafusion::LiquidParquetSource; -use logforth::filter::env_filter::EnvFilterBuilder; +use logforth::filter::rustlog::RustLogFilterBuilder; use opentelemetry::InstrumentationScope; use opentelemetry::KeyValue; use opentelemetry_otlp::SpanExporter; @@ -54,7 +54,7 @@ pub fn instrument_liquid_source_with_span( pub fn setup_observability(service_name: &str, jaeger_endpoint: Option<&str>) { logforth::starter_log::builder() .dispatch(|d| { - d.filter(EnvFilterBuilder::from_default_env().build()) + d.filter(RustLogFilterBuilder::from_default_env().build()) .append(logforth::append::Stdout::default()) }) .apply(); diff --git a/dev/README.md b/dev/README.md index 72aa3c031..8257cb6b8 100644 --- a/dev/README.md +++ b/dev/README.md @@ -9,6 +9,12 @@ curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh ``` +Alternatively, use the Nix dev shell (works on both Linux and macOS; also provides the tooling for `dev-tools`, e.g. dioxus-cli, tailwindcss, wasm-bindgen): + +```bash +nix develop +``` + Run tests: ```bash diff --git a/dev/dev-tools/Cargo.toml b/dev/dev-tools/Cargo.toml index 879412d3e..9e8824cb9 100644 --- a/dev/dev-tools/Cargo.toml +++ b/dev/dev-tools/Cargo.toml @@ -3,10 +3,17 @@ name = "dev-tools" version = "0.1.3" authors = ["XiangpengHao "] edition = "2024" +rust-version = { workspace = true } [dependencies] -dioxus = { version = "=0.7.5", features = ["router", "fullstack"] } +dioxus = { version = "=0.7.10", features = ["router", "fullstack"] } +# Must match the wasm-bindgen-cli version pinned in flake.nix. +# Only used through dioxus macros, hence the cargo-shear ignore. +wasm-bindgen = "=0.2.126" + +[package.metadata.cargo-shear] +ignored = ["wasm-bindgen"] [features] default = ["web"] diff --git a/examples/Cargo.toml b/examples/Cargo.toml index 36b1704ef..fc8ecdb39 100644 --- a/examples/Cargo.toml +++ b/examples/Cargo.toml @@ -1,6 +1,7 @@ [package] name = "examples" edition = { workspace = true } +rust-version = { workspace = true } publish = false [[bin]] diff --git a/flake.lock b/flake.lock index 91d09a14a..3a7310330 100644 --- a/flake.lock +++ b/flake.lock @@ -20,11 +20,11 @@ }, "nixpkgs": { "locked": { - "lastModified": 1777268161, - "narHash": "sha256-bxrdOn8SCOv8tN4JbTF/TXq7kjo9ag4M+C8yzzIRYbE=", + "lastModified": 1788039129, + "narHash": "sha256-pa4Q0qErvCvzCaaUph7Sm37RhR4xvPrYI8Lgz6k85+A=", "owner": "NixOS", "repo": "nixpkgs", - "rev": "1c3fe55ad329cbcb28471bb30f05c9827f724c76", + "rev": "d2f67949798825fe853f7c5d0492b8bf016d3f88", "type": "github" }, "original": { @@ -62,11 +62,11 @@ "nixpkgs": "nixpkgs_2" }, "locked": { - "lastModified": 1777432579, - "narHash": "sha256-Ce11TStDsqCge2vAAfLKe2+4lDI5cSX5ZYZOuKJBKKQ=", + "lastModified": 1788165049, + "narHash": "sha256-en4IoUeCqvq9F66YhwOrUFw1nc70OBhrJwrzfalezvY=", "owner": "oxalica", "repo": "rust-overlay", - "rev": "3ecb5e6ab380ced3272ef7fcfe398bffbcc0f152", + "rev": "d03cd474bd97389dcc2e8cd3b3bb6b8c6e346b1a", "type": "github" }, "original": { diff --git a/flake.nix b/flake.nix index 576890697..c54fc1ed0 100644 --- a/flake.nix +++ b/flake.nix @@ -47,16 +47,16 @@ nodejs tailwindcss_4 dioxus-cli - wasm-bindgen-cli_0_2_118 + wasm-bindgen-cli_0_2_126 binaryen (rust-bin.selectLatestNightlyWith (toolchain: toolchain.default.override { extensions = [ "rust-src" "llvm-tools-preview" ]; - targets = [ "x86_64-unknown-linux-gnu" "wasm32-unknown-unknown" ]; + targets = [ "wasm32-unknown-unknown" ]; })) ] # perf and bpftrace exist only on Linux in nixpkgs, and this flake # is evaluated for every default system, macOS included. - ++ lib.optionals stdenv.isLinux [ + ++ lib.optionals stdenv.hostPlatform.isLinux [ bpftrace perf ]; diff --git a/fuzz/Cargo.toml b/fuzz/Cargo.toml index 1ae1a25ce..53dc3d5f4 100644 --- a/fuzz/Cargo.toml +++ b/fuzz/Cargo.toml @@ -3,6 +3,7 @@ name = "liquid-cache-fuzz" version = "0.1.13" publish = false edition = "2024" +rust-version = { workspace = true } [package.metadata] cargo-fuzz = true diff --git a/src/common/Cargo.toml b/src/common/Cargo.toml index 582f4bc9c..8bca5f6ab 100644 --- a/src/common/Cargo.toml +++ b/src/common/Cargo.toml @@ -2,6 +2,7 @@ name = "liquid-cache-common" version = { workspace = true } edition = { workspace = true } +rust-version = { workspace = true } license = { workspace = true } readme = "README.md" description = { workspace = true } diff --git a/src/core/Cargo.toml b/src/core/Cargo.toml index d8e2b9f49..af15ed62d 100644 --- a/src/core/Cargo.toml +++ b/src/core/Cargo.toml @@ -2,6 +2,7 @@ name = "liquid-cache" version = { workspace = true } edition = { workspace = true } +rust-version = { workspace = true } license = { workspace = true } readme = "README.md" description = { workspace = true } @@ -21,9 +22,9 @@ datafusion-physical-expr = { workspace = true } datafusion-physical-expr-common = { workspace = true } arrow = { workspace = true } arrow-schema = { workspace = true } -fastlanes = "0.5.1" +fastlanes = "0.7.0" num-traits = "0.2.19" -fsst-rs = "0.5.11" +fsst-rs = "0.6.0" ahash = { workspace = true } tempfile = { workspace = true } congee = { workspace = true } @@ -34,7 +35,7 @@ parquet-variant-compute = { workspace = true } fastrace = { workspace = true } serde = { workspace = true } serde_json = { workspace = true } -sysinfo = { version = "0.38.4", default-features = false, features = ["system"] } +sysinfo = { version = "0.39.6", default-features = false, features = ["system"] } [dev-dependencies] tempfile = { workspace = true } diff --git a/src/core/README.md b/src/core/README.md index d263aab51..1338bc2e6 100644 --- a/src/core/README.md +++ b/src/core/README.md @@ -27,7 +27,7 @@ let arrow_array = Arc::new(UInt64Array::from_iter_values(0..1000)); // Insert once; replacement/placement is handled by the cache policy storage.insert(entry_id, arrow_array.clone()).await; -assert!(storage.is_cached(&entry_id)); +assert!(storage.contains(&entry_id)); }); ``` diff --git a/src/core/src/cache/builders.rs b/src/core/src/cache/builders.rs index af0572dac..78e8300b3 100644 --- a/src/core/src/cache/builders.rs +++ b/src/core/src/cache/builders.rs @@ -269,7 +269,6 @@ impl<'a> Get<'a> { /// Materialize the cached array as [`ArrayRef`]. pub async fn read(self) -> Option { - self.storage.observer().on_get(self.selection.is_some()); self.storage .read_arrow_array( self.entry_id, diff --git a/src/core/src/cache/core.rs b/src/core/src/cache/core.rs index 2f85a1a7d..8aee66d2b 100644 --- a/src/core/src/cache/core.rs +++ b/src/core/src/cache/core.rs @@ -113,11 +113,21 @@ pub struct LiquidCache { disk_copies: Mutex>, } +/// Outcome of [`LiquidCache::prefetch`]. +pub enum PrefetchResult { + /// A memory-form snapshot of the entry (Arrow or Liquid), ready to hand to a reader. + Snapshot(Arc), + /// The entry is squeezed; prefetch leaves it alone. + Squeezed, + /// The entry is not in the index, or its disk blob is gone. + Absent, +} + /// Builder returned by [`LiquidCache::insert`] for configuring cache writes. impl LiquidCache { /// Return current cache statistics: counts and resource usage. pub fn stats(&self) -> CacheStats { - // Count entries by residency/format + // Count entries by storage tier and format let total_entries = self.index.entry_count(); let mut memory_arrow_entries = 0usize; @@ -192,6 +202,35 @@ impl LiquidCache { EvaluatePredicate::new(self, entry_id, predicate) } + /// Prefetch an entry into a memory-form snapshot without recording an access. + pub async fn prefetch(&self, entry_id: &EntryID) -> PrefetchResult { + let Some(entry) = self.index.get(entry_id) else { + return PrefetchResult::Absent; + }; + match entry.as_ref() { + CacheEntry::MemoryArrow(_) | CacheEntry::MemoryLiquid(_) => { + PrefetchResult::Snapshot(entry) + } + disk @ CacheEntry::DiskArrow { .. } => { + let Some(array) = self.read_disk_arrow_array(entry_id).await else { + return PrefetchResult::Absent; + }; + self.maybe_hydrate(entry_id, disk, MaterializedEntry::Arrow(&array), None) + .await; + PrefetchResult::Snapshot(Arc::new(CacheEntry::memory_arrow(array))) + } + disk @ CacheEntry::DiskLiquid { .. } => { + let Some(array) = self.read_disk_liquid_array(entry_id).await else { + return PrefetchResult::Absent; + }; + self.maybe_hydrate(entry_id, disk, MaterializedEntry::Liquid(&array), None) + .await; + PrefetchResult::Snapshot(Arc::new(CacheEntry::memory_liquid(array))) + } + CacheEntry::MemorySqueezedLiquid(_) => PrefetchResult::Squeezed, + } + } + /// Try to read a liquid array from the cache. /// Returns None if the cached data is not in liquid format. pub async fn try_read_liquid( @@ -207,14 +246,14 @@ impl LiquidCache { match batch.as_ref() { CacheEntry::MemoryLiquid(array) => Some(array.clone()), entry @ CacheEntry::DiskLiquid { .. } => { - let liquid = self.read_disk_liquid_array(entry_id).await; + let liquid = self.read_disk_liquid_array(entry_id).await?; self.maybe_hydrate(entry_id, entry, MaterializedEntry::Liquid(&liquid), None) .await; Some(liquid) } CacheEntry::MemorySqueezedLiquid(array) => match array.disk_backing() { SqueezedBacking::Liquid(_) => { - let liquid = self.read_disk_liquid_array(entry_id).await; + let liquid = self.read_disk_liquid_array(entry_id).await?; Some(liquid) } SqueezedBacking::Arrow(_) => None, @@ -237,9 +276,9 @@ impl LiquidCache { self.disk_copies.lock().unwrap().clear(); } - /// Check if a batch is cached. - pub fn is_cached(&self, entry_id: &EntryID) -> bool { - self.index.is_cached(entry_id) + /// Check whether the cache contains a batch. + pub fn contains(&self, entry_id: &EntryID) -> bool { + self.index.contains(entry_id) } /// Get the config of the cache. @@ -776,19 +815,44 @@ impl LiquidCache { selection: Option<&BooleanBuffer>, expression: Option<&CacheExpression>, ) -> Option { - use arrow::array::BooleanArray; - + self.observer.on_get(selection.is_some()); let batch = self.index.get(entry_id)?; self.cache_policy .notify_access(entry_id, CachedBatchType::from(batch.as_ref())); + self.read_entry_inner(entry_id, batch.as_ref(), selection, expression) + .await + } + + /// Read an already-looked-up cache entry. + pub async fn read_entry( + &self, + entry_id: &EntryID, + entry: &CacheEntry, + selection: Option<&BooleanBuffer>, + expression: Option<&CacheExpression>, + ) -> Option { + self.observer.on_get(selection.is_some()); + self.read_entry_inner(entry_id, entry, selection, expression) + .await + } + + async fn read_entry_inner( + &self, + entry_id: &EntryID, + entry: &CacheEntry, + selection: Option<&BooleanBuffer>, + expression: Option<&CacheExpression>, + ) -> Option { + use arrow::array::BooleanArray; + self.trace(InternalEvent::Read { entry: *entry_id, selection: selection.is_some(), expr: expression.cloned(), - cached: CachedBatchType::from(batch.as_ref()), + cached: CachedBatchType::from(entry), }); - match batch.as_ref() { + match entry { CacheEntry::MemoryArrow(array) => match selection { Some(selection) => { let selection_array = BooleanArray::new(selection.clone(), None); @@ -801,7 +865,7 @@ impl LiquidCache { None => Some(array.to_arrow_array()), }, CacheEntry::DiskArrow { .. } | CacheEntry::DiskLiquid { .. } => { - self.read_disk_array(batch.as_ref(), entry_id, expression, selection) + self.read_disk_array(entry, entry_id, expression, selection) .await } CacheEntry::MemorySqueezedLiquid(array) => { @@ -825,7 +889,7 @@ impl LiquidCache { { return Some(arrow::array::new_empty_array(data_type)); } - let full_array = self.read_disk_arrow_array(entry_id).await; + let full_array = self.read_disk_arrow_array(entry_id).await?; self.maybe_hydrate( entry_id, entry, @@ -847,7 +911,7 @@ impl LiquidCache { { return Some(arrow::array::new_empty_array(data_type)); } - let liquid = self.read_disk_liquid_array(entry_id).await; + let liquid = self.read_disk_liquid_array(entry_id).await?; self.maybe_hydrate( entry_id, entry, @@ -940,7 +1004,7 @@ impl LiquidCache { let full_array = if !all_paths_present { let batch = CacheEntry::MemorySqueezedLiquid(array.clone()); self.observer.on_get_squeezed_needs_io(); - let full_array = self.read_disk_arrow_array(entry_id).await; + let full_array = self.read_disk_arrow_array(entry_id).await?; self.maybe_hydrate( entry_id, &batch, @@ -1016,12 +1080,12 @@ impl LiquidCache { Ok(()) } - async fn read_disk_arrow_array(&self, entry_id: &EntryID) -> ArrayRef { - let bytes = self - .store - .get(&entry_id_to_key(entry_id)) - .await - .expect("read failed"); + async fn read_disk_arrow_array(&self, entry_id: &EntryID) -> Option { + let bytes = match self.store.get(&entry_id_to_key(entry_id)).await { + Ok(bytes) => bytes, + Err(t4::Error::NotFound) => return None, + Err(error) => panic!("read failed: {error}"), + }; let bytes_len = bytes.len(); let cursor = std::io::Cursor::new(bytes); let mut reader = @@ -1032,18 +1096,18 @@ impl LiquidCache { entry: *entry_id, bytes: bytes_len, }); - array + Some(array) } async fn read_disk_liquid_array( &self, entry_id: &EntryID, - ) -> crate::liquid_array::LiquidArrayRef { - let bytes = self - .store - .get(&entry_id_to_key(entry_id)) - .await - .expect("read failed"); + ) -> Option { + let bytes = match self.store.get(&entry_id_to_key(entry_id)).await { + Ok(bytes) => bytes, + Err(t4::Error::NotFound) => return None, + Err(error) => panic!("read failed: {error}"), + }; self.trace(InternalEvent::IoReadLiquid { entry: *entry_id, bytes: bytes.len(), @@ -1051,10 +1115,12 @@ impl LiquidCache { let compressor_states = self.metadata.get_compressor(entry_id); let compressor = compressor_states.fsst_compressor(); - (crate::liquid_array::ipc::read_from_bytes( - Bytes::from(bytes), - &crate::liquid_array::ipc::LiquidIPCContext::new(compressor), - )) as _ + Some( + (crate::liquid_array::ipc::read_from_bytes( + Bytes::from(bytes), + &crate::liquid_array::ipc::LiquidIPCContext::new(compressor), + )) as _, + ) } pub(crate) async fn eval_predicate_internal( @@ -1063,19 +1129,41 @@ impl LiquidCache { selection_opt: Option<&BooleanBuffer>, predicate: &LiquidExpr, ) -> Option { - use arrow::array::BooleanArray; - self.observer.on_eval_predicate(); let batch = self.index.get(entry_id)?; self.cache_policy .notify_access(entry_id, CachedBatchType::from(batch.as_ref())); + self.eval_predicate_on_entry_inner(entry_id, batch.as_ref(), selection_opt, predicate) + .await + } + + /// Evaluate a predicate on an already-looked-up cache entry. + pub async fn eval_predicate_on_entry( + &self, + entry_id: &EntryID, + entry: &CacheEntry, + selection_opt: Option<&BooleanBuffer>, + predicate: &LiquidExpr, + ) -> Option { + self.observer.on_eval_predicate(); + self.eval_predicate_on_entry_inner(entry_id, entry, selection_opt, predicate) + .await + } + + async fn eval_predicate_on_entry_inner( + &self, + entry_id: &EntryID, + entry: &CacheEntry, + selection_opt: Option<&BooleanBuffer>, + predicate: &LiquidExpr, + ) -> Option { self.trace(InternalEvent::EvalPredicate { entry: *entry_id, selection: selection_opt.is_some(), - cached: CachedBatchType::from(batch.as_ref()), + cached: CachedBatchType::from(entry), }); - match batch.as_ref() { + match entry { CacheEntry::MemoryArrow(array) => { let mut owned = None; let selection = selection_opt.unwrap_or_else(|| { @@ -1088,7 +1176,7 @@ impl LiquidCache { Some(self.eval_predicate_on_array(filtered, predicate)) } entry @ CacheEntry::DiskArrow { .. } => { - let array = self.read_disk_arrow_array(entry_id).await; + let array = self.read_disk_arrow_array(entry_id).await?; self.maybe_hydrate(entry_id, entry, MaterializedEntry::Arrow(&array), None) .await; let mut owned = None; @@ -1110,7 +1198,7 @@ impl LiquidCache { Some(array.try_eval_predicate(predicate, selection)) } entry @ CacheEntry::DiskLiquid { .. } => { - let liquid = self.read_disk_liquid_array(entry_id).await; + let liquid = self.read_disk_liquid_array(entry_id).await?; self.maybe_hydrate(entry_id, entry, MaterializedEntry::Liquid(&liquid), None) .await; let mut owned = None; @@ -1427,6 +1515,25 @@ mod tests { } } + #[tokio::test] + async fn missing_disk_blob_is_a_cache_miss() { + let directory = tempfile::tempdir().unwrap(); + let store = crate::store::mount(directory.path().join("cache.t4")) + .await + .unwrap(); + let cache = LiquidCacheBuilder::new() + .with_store(store.clone()) + .build() + .await; + let id = EntryID::from(320usize); + + cache.insert(id, create_test_arrow_array(8)).await.unwrap(); + cache.flush_all_to_disk().await.unwrap(); + store.remove(&entry_id_to_key(&id)).await.unwrap(); + + assert!(cache.get(&id).await.is_none()); + } + #[tokio::test] async fn hydrate_disk_liquid_on_get_promotes_to_memory_liquid() { let store = create_cache_store(1 << 20, Box::new(LiquidPolicy::new())).await; @@ -1466,7 +1573,7 @@ mod tests { let err = cache.insert(EntryID::from(900usize), array).await; assert_eq!(err, Err(CacheFull)); - assert!(!cache.is_cached(&EntryID::from(900usize))); + assert!(!cache.contains(&EntryID::from(900usize))); } #[tokio::test] @@ -1487,12 +1594,12 @@ mod tests { let second = EntryID::from(911usize); cache.insert(first, first_array).await.unwrap(); cache.flush_all_to_disk().await.unwrap(); - assert!(cache.is_cached(&first)); + assert!(cache.contains(&first)); cache.insert(second, second_array).await.unwrap(); cache.flush_all_to_disk().await.unwrap(); - assert!(!cache.is_cached(&first)); + assert!(!cache.contains(&first)); assert!(matches!( cache.index().get(&second).unwrap().as_ref(), CacheEntry::DiskArrow { .. } @@ -1519,7 +1626,7 @@ mod tests { cache.flush_all_to_disk().await.unwrap(); - assert!(!cache.is_cached(&first) || !cache.is_cached(&second)); + assert!(!cache.contains(&first) || !cache.contains(&second)); } #[tokio::test] @@ -1541,7 +1648,7 @@ mod tests { cache.remove_disk_entry(entry).await; assert_eq!(cache.stats().disk_usage_bytes, before - disk_bytes); - assert!(!cache.is_cached(&entry)); + assert!(!cache.contains(&entry)); } #[tokio::test] @@ -1559,7 +1666,7 @@ mod tests { let result = cache.flush_all_to_disk().await; assert_eq!(result, Ok(())); - assert!(!cache.is_cached(&entry_id)); + assert!(!cache.contains(&entry_id)); } async fn hydrating_cache() -> Arc { @@ -1800,7 +1907,7 @@ mod tests { let result = cache.insert(id, too_big).await; assert_eq!(result, Err(CacheFull)); - assert!(!cache.is_cached(&id)); + assert!(!cache.contains(&id)); assert!(cache.get(&id).await.is_none()); assert_eq!(cache.budget().disk_usage_bytes(), 0); assert_eq!(cache.budget().memory_usage_bytes(), 0); @@ -1837,7 +1944,7 @@ mod tests { // that is full with the entry's own copy, so the entry is dropped. cache.flush_all_to_disk().await.unwrap(); - assert!(!cache.is_cached(&id)); + assert!(!cache.contains(&id)); assert_eq!( cache.budget().disk_usage_bytes(), 0, diff --git a/src/core/src/cache/index.rs b/src/core/src/cache/index.rs index cf8881ab5..89adce196 100644 --- a/src/core/src/cache/index.rs +++ b/src/core/src/cache/index.rs @@ -69,7 +69,9 @@ impl ArtIndex { self.art.get(*entry_id, &guard)?.load() } - pub(crate) fn is_cached(&self, entry_id: &EntryID) -> bool { + // Delegates to `get`: a slot can outlive its entry across a remove or a + // replace, so slot presence alone would report an absent entry as cached. + pub(crate) fn contains(&self, entry_id: &EntryID) -> bool { self.get(entry_id).is_some() } @@ -127,15 +129,15 @@ mod tests { use super::*; #[test] - fn test_get_and_is_cached() { + fn test_get_and_contains() { let store = ArtIndex::new(); let entry_id1: EntryID = EntryID::from(1); let entry_id2: EntryID = EntryID::from(2); let array1 = create_test_array(100); // Initially, entries should not be cached - assert!(!store.is_cached(&entry_id1)); - assert!(!store.is_cached(&entry_id2)); + assert!(!store.contains(&entry_id1)); + assert!(!store.contains(&entry_id2)); assert!(store.get(&entry_id1).is_none()); // Insert an entry and verify it's cached @@ -143,8 +145,8 @@ mod tests { store.insert(&entry_id1, array1.clone()); } - assert!(store.is_cached(&entry_id1)); - assert!(!store.is_cached(&entry_id2)); + assert!(store.contains(&entry_id1)); + assert!(!store.contains(&entry_id2)); // Get should return the cached value match store.get(&entry_id1) { @@ -165,11 +167,11 @@ mod tests { store.insert(&entry_id, array.clone()); let entry_id: EntryID = EntryID::from(1); - assert!(store.is_cached(&entry_id)); + assert!(store.contains(&entry_id)); store.reset(); let entry_id: EntryID = EntryID::from(1); - assert!(!store.is_cached(&entry_id)); + assert!(!store.contains(&entry_id)); } /// The array behind a removed or replaced entry must die with the last diff --git a/src/core/src/cache/mod.rs b/src/core/src/cache/mod.rs index 4f52abe56..19ae5dc13 100644 --- a/src/core/src/cache/mod.rs +++ b/src/core/src/cache/mod.rs @@ -15,7 +15,7 @@ mod utils; pub use builders::{EvaluatePredicate, Get, Insert, LiquidCacheBuilder, default_max_memory_bytes}; pub use cached_batch::{CacheEntry, CachedBatchType}; -pub use core::LiquidCache; +pub use core::{LiquidCache, PrefetchResult}; pub use expressions::{CacheExpression, VariantRequest}; #[cfg(test)] pub(crate) use io_context::TestSqueezeIo; diff --git a/src/core/src/cache/policies/squeeze.rs b/src/core/src/cache/policies/squeeze.rs index fd9a2e511..7941a1feb 100644 --- a/src/core/src/cache/policies/squeeze.rs +++ b/src/core/src/cache/policies/squeeze.rs @@ -676,9 +676,7 @@ mod tests { inner .column_by_name("metadata") .cloned() - .unwrap_or_else(|| { - Arc::new(base_variant.metadata_column().clone()) as ArrayRef - }), + .unwrap_or_else(|| base_variant.metadata_column().clone()), inner.column_by_name("value").cloned().unwrap_or_else(|| { Arc::new(BinaryViewArray::from(vec![None::<&[u8]>; inner.len()])) as ArrayRef }), diff --git a/src/core/src/liquid_array/byte_view_array/comparisons.rs b/src/core/src/liquid_array/byte_view_array/comparisons.rs index 2fd21e6d2..38b5efc7d 100644 --- a/src/core/src/liquid_array/byte_view_array/comparisons.rs +++ b/src/core/src/liquid_array/byte_view_array/comparisons.rs @@ -540,8 +540,10 @@ fn compare_with_arrow_inner( fn compress_needle(compressor: &Compressor, needle: &[u8]) -> Vec { let mut compressed = Vec::with_capacity(needle.len().saturating_mul(2)); + // SAFETY: the largest compressed size is all escapes == 2 * plaintext_len. unsafe { - compressor.compress_into(needle, &mut compressed); + let len = compressor.compress_into(needle, compressed.spare_capacity_mut()); + compressed.set_len(len); } compressed } diff --git a/src/core/src/liquid_array/raw/fsst_buffer.rs b/src/core/src/liquid_array/raw/fsst_buffer.rs index 4e64b6f4d..0789623ff 100644 --- a/src/core/src/liquid_array/raw/fsst_buffer.rs +++ b/src/core/src/liquid_array/raw/fsst_buffer.rs @@ -70,7 +70,8 @@ impl RawFsstBuffer { // (all bytes escaped) which is `2 * plaintext_len`. compress_buffer.reserve(bytes.len().saturating_mul(2)); unsafe { - compressor.compress_into(bytes, compress_buffer); + let len = compressor.compress_into(bytes, compress_buffer.spare_capacity_mut()); + compress_buffer.set_len(len); } values_buffer.extend_from_slice(compress_buffer); diff --git a/src/core/study/fsst_selectivity.rs b/src/core/study/fsst_selectivity.rs index ed349858e..9e0ae16e4 100644 --- a/src/core/study/fsst_selectivity.rs +++ b/src/core/study/fsst_selectivity.rs @@ -99,7 +99,7 @@ async fn main() { continue; } - // Warm up once to reduce cold-start noise. + // Run once to reduce cold-start noise. std::hint::black_box(fsst.to_uncompressed_selected(&selection.indices)); let mut total = 0.0; diff --git a/src/datafusion-client/Cargo.toml b/src/datafusion-client/Cargo.toml index ddd207cd1..f8862d718 100644 --- a/src/datafusion-client/Cargo.toml +++ b/src/datafusion-client/Cargo.toml @@ -2,6 +2,7 @@ name = "liquid-cache-datafusion-client" authors = { workspace = true } edition = { workspace = true } +rust-version = { workspace = true } version = { workspace = true } license = { workspace = true } readme = "README.md" diff --git a/src/datafusion-client/src/client_exec.rs b/src/datafusion-client/src/client_exec.rs index 76b63c441..13ee28f55 100644 --- a/src/datafusion-client/src/client_exec.rs +++ b/src/datafusion-client/src/client_exec.rs @@ -23,7 +23,10 @@ use datafusion::physical_plan::filter_pushdown::{ ChildPushdownResult, FilterDescription, FilterPushdownPhase, FilterPushdownPropagation, }; use datafusion::physical_plan::metrics::{ExecutionPlanMetricsSet, MetricsSet}; -use datafusion::physical_plan::{ExecutionPlanProperties, PhysicalExpr, PlanProperties}; +use datafusion::physical_plan::{ + ChildrenPropertiesMode, ExecutionPlanProperties, PhysicalExpr, PlanProperties, + ReplaceChildrenOptions, +}; use datafusion::{ error::Result, execution::{RecordBatchStream, SendableRecordBatchStream}, @@ -75,18 +78,22 @@ impl std::fmt::Debug for LiquidCacheClientExec { } impl LiquidCacheClientExec { + fn plan_properties(remote_plan: &Arc) -> Arc { + Arc::new(PlanProperties::new( + remote_plan.equivalence_properties().clone(), + remote_plan.output_partitioning().clone(), + remote_plan.pipeline_behavior(), + remote_plan.boundedness(), + )) + } + pub(crate) fn new( remote_plan: Arc, cache_server: String, object_stores: Vec<(ObjectStoreUrl, HashMap)>, squeeze_hints: ColumnSqueezeHints, ) -> Self { - let properties = Arc::new(PlanProperties::new( - remote_plan.equivalence_properties().clone(), // Equivalence Properties - remote_plan.output_partitioning().clone(), // Output Partitioning - remote_plan.pipeline_behavior(), - remote_plan.boundedness(), - )); + let properties = Self::plan_properties(&remote_plan); let uuid = Uuid::new_v4(); Self { remote_plan, @@ -140,31 +147,51 @@ impl ExecutionPlan for LiquidCacheClientExec { vec![&self.remote_plan] } - /// The client node holds no expressions of its own; the wrapped remote plan - /// is visited as a child. - fn apply_expressions( - &self, - _f: &mut dyn FnMut(&Arc) -> Result, - ) -> Result { - Ok(TreeNodeRecursion::Continue) - } - - fn with_new_children( + fn replace_children( self: Arc, - children: Vec>, + mut children: Vec>, + options: ReplaceChildrenOptions, ) -> datafusion::error::Result> { + if children.len() != 1 { + return internal_err!( + "LiquidCacheClientExec expects one child, received {}", + children.len() + ); + } + let remote_plan = children.swap_remove(0); + let properties = match options.children_properties { + ChildrenPropertiesMode::Keep => Arc::clone(&self.properties), + ChildrenPropertiesMode::Recompute => Self::plan_properties(&remote_plan), + }; Ok(Arc::new(Self { - remote_plan: children.first().unwrap().clone(), + remote_plan, cache_server: self.cache_server.clone(), plan_registered: self.plan_registered.clone(), object_stores: self.object_stores.clone(), metrics: self.metrics.clone(), uuid: self.uuid, - properties: self.properties.clone(), + properties, squeeze_hints: self.squeeze_hints.clone(), })) } + fn with_new_children( + self: Arc, + children: Vec>, + ) -> datafusion::error::Result> { + self.replace_children( + children, + ReplaceChildrenOptions::new(ChildrenPropertiesMode::Recompute), + ) + } + + fn apply_expressions( + &self, + _f: &mut dyn FnMut(&Arc) -> Result, + ) -> Result { + Ok(TreeNodeRecursion::Continue) + } + fn execute( &self, partition: usize, diff --git a/src/datafusion-client/src/optimizer.rs b/src/datafusion-client/src/optimizer.rs index a7a406427..6a3ca6c6c 100644 --- a/src/datafusion-client/src/optimizer.rs +++ b/src/datafusion-client/src/optimizer.rs @@ -1,11 +1,15 @@ use std::{collections::HashMap, sync::Arc}; use datafusion::{ - config::ConfigOptions, datasource::source::DataSourceExec, error::Result, - execution::object_store::ObjectStoreUrl, physical_optimizer::PhysicalOptimizerRule, - physical_plan::ExecutionPlan, physical_plan::aggregates::AggregateExec, - physical_plan::aggregates::AggregateMode, physical_plan::repartition::RepartitionExec, - physical_plan::replace_children_if_necessary, + config::ConfigOptions, + datasource::source::DataSourceExec, + error::Result, + execution::object_store::ObjectStoreUrl, + physical_optimizer::PhysicalOptimizerRule, + physical_plan::aggregates::AggregateExec, + physical_plan::aggregates::AggregateMode, + physical_plan::repartition::RepartitionExec, + physical_plan::{ExecutionPlan, execution_plan::replace_children_if_necessary}, }; use liquid_cache_datafusion::optimizers::SqueezeHintMap; diff --git a/src/datafusion-local/Cargo.toml b/src/datafusion-local/Cargo.toml index f7687f16c..13a14dc3c 100644 --- a/src/datafusion-local/Cargo.toml +++ b/src/datafusion-local/Cargo.toml @@ -2,6 +2,7 @@ name = "liquid-cache-datafusion-local" version = { workspace = true } edition = { workspace = true } +rust-version = { workspace = true } license = { workspace = true } readme = "README.md" description = { workspace = true } diff --git a/src/datafusion-local/src/lib.rs b/src/datafusion-local/src/lib.rs index 27254ea73..e0474844b 100644 --- a/src/datafusion-local/src/lib.rs +++ b/src/datafusion-local/src/lib.rs @@ -7,10 +7,9 @@ mod tests; use std::path::PathBuf; use std::sync::Arc; -use datafusion::common::config::ConfigNonZeroUsize; -use datafusion::error::Result; use datafusion::logical_expr::ScalarUDF; use datafusion::prelude::{SessionConfig, SessionContext}; +use datafusion::{common::config::ConfigNonZeroUsize, error::Result}; use liquid_cache::cache::squeeze_policies::{SqueezePolicy, TranscodeSqueezeEvict}; use liquid_cache::cache::{AlwaysHydrate, HydrationPolicy, default_max_memory_bytes}; use liquid_cache::cache_policies::{CachePolicy, LiquidPolicy}; @@ -74,6 +73,7 @@ pub struct LiquidCacheLocalBuilder { /// When set, a scan is cached only if its estimated liquid footprint stays /// within `budget × tolerance`; `strict` toggles fail-loud panic handling. admission: Option<(f64, f64, f64, bool)>, + prefetch: bool, span: fastrace::Span, } @@ -90,6 +90,7 @@ impl Default for LiquidCacheLocalBuilder { squeeze_policy: Box::new(TranscodeSqueezeEvict), hydration_policy: Box::new(AlwaysHydrate::new()), admission: None, + prefetch: true, span: fastrace::Span::enter_with_local_parent("liquid_cache_datafusion_local_builder"), } } @@ -145,6 +146,12 @@ impl LiquidCacheLocalBuilder { self } + /// Enable or disable row-group prefetching. + pub fn with_prefetch(mut self, prefetch: bool) -> Self { + self.prefetch = prefetch; + self + } + /// Set fastrace span pub fn with_span(mut self, span: fastrace::Span) -> Self { self.span = span; @@ -215,7 +222,7 @@ impl LiquidCacheLocalBuilder { .await; let cache_ref = Arc::new(cache); - let mut optimizer = LocalModeOptimizer::new(cache_ref.clone()); + let mut optimizer = LocalModeOptimizer::new(cache_ref.clone()).with_prefetch(self.prefetch); if let Some((expansion, safety, tolerance, strict)) = self.admission { optimizer = optimizer.with_admission_gate(expansion, safety, tolerance, strict); } diff --git a/src/datafusion-local/src/tests/mod.rs b/src/datafusion-local/src/tests/mod.rs index ec3f750f1..6a9f04062 100644 --- a/src/datafusion-local/src/tests/mod.rs +++ b/src/datafusion-local/src/tests/mod.rs @@ -26,6 +26,7 @@ mod batch_size_alignment; mod column_free_conjunct; mod date_optimizer; mod filter_limit; +mod nested_filter; mod page_index; mod squeeze; mod unevaluable_conjunct; @@ -121,9 +122,14 @@ async fn create_session_context_with_liquid_cache( cache_size_bytes: usize, cache_dir: &Path, ) -> Result<(SessionContext, LiquidCacheParquetRef)> { - let mut config = cache_test_config(); + // These tests snapshot exact cache contents and counters. A repartitioned + // file scan populates the cache concurrently, so insertion order (and, for + // LIMIT queries, which partitions finish before cancellation) is not a + // stable property to snapshot. + let mut config = SessionConfig::new().with_repartition_file_scans(false); config.options_mut().execution.target_partitions = 4; let (ctx, cache) = LiquidCacheLocalBuilder::new() + .with_prefetch(false) .with_max_memory_bytes(cache_size_bytes) .with_cache_dir(cache_dir.to_path_buf()) .with_squeeze_policy(squeeze_policy) @@ -145,6 +151,54 @@ async fn get_physical_plan(sql: &str, ctx: &SessionContext) -> Arc String { + let plan = get_physical_plan(sql, ctx).await; + let batches = collect(plan, ctx.task_ctx()).await.unwrap(); + pretty_format_batches(&batches).unwrap().to_string() +} + +async fn run_io_profile(prefetch: bool, cache_dir: &Path) -> (String, u64, u64, u64) { + let config = SessionConfig::new().with_repartition_file_scans(false); + let builder = LiquidCacheLocalBuilder::new() + .with_max_memory_bytes(64 * 1024 * 1024) + .with_cache_dir(cache_dir.to_path_buf()); + let builder = if prefetch { + builder + } else { + builder.with_prefetch(false) + }; + let (ctx, cache) = builder.build(config).await.unwrap(); + ctx.register_parquet("hits", TEST_FILE, ParquetReadOptions::default()) + .await + .unwrap(); + let sql = r#"SELECT "WatchID" FROM hits WHERE "SearchPhrase" LIKE '%abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMNOPQRSTUVWXYZ0123456789%'"#; + + let first = get_result(&ctx, sql).await; + cache.flush_data().await.unwrap(); + cache.storage().stats(); + let second = get_result(&ctx, sql).await; + let runtime = cache.storage().stats().runtime; + assert_eq!(first, second); + + ( + second, + runtime.read_io_count, + runtime.get, + runtime.eval_predicate, + ) +} + +#[tokio::test] +async fn prefetch_matches_lazy_io() { + let lazy_dir = TempDir::new().unwrap(); + let prefetch_dir = TempDir::new().unwrap(); + + let lazy = run_io_profile(false, lazy_dir.path()).await; + let prefetch = run_io_profile(true, prefetch_dir.path()).await; + + assert_eq!(lazy, prefetch); +} + async fn run_sql_with_cache( sql: &str, squeeze_policy: Box, @@ -160,13 +214,7 @@ async fn run_sql_with_cache( let displayable = DisplayableExecutionPlan::new(plan.as_ref()); let plan_string = format!("{}", displayable.tree_render()); - async fn get_result(ctx: &SessionContext, sql: &str) -> String { - let plan = get_physical_plan(sql, ctx).await; - let batches = collect(plan, ctx.task_ctx()).await.unwrap(); - pretty_format_batches(&batches).unwrap().to_string() - } - - // Clear any historical runtime counters before warming the cache. + // Clear any historical runtime counters before prefetching the cache. cache.storage().stats(); let first_run = get_result(&ctx, sql).await; @@ -405,6 +453,7 @@ async fn test_provide_schema2() { let mut config = cache_test_config(); config.options_mut().execution.target_partitions = 4; let (liquid_ctx, cache) = LiquidCacheLocalBuilder::new() + .with_prefetch(false) .with_cache_dir(cache_dir.path().to_path_buf()) .with_max_memory_bytes(1024 * 1024) .with_squeeze_policy(Box::new(TranscodeSqueezeEvict)) @@ -453,7 +502,7 @@ async fn test_provide_schema2() { let displayable = DisplayableExecutionPlan::new(plan.as_ref()); let plan_string = format!("{}", displayable.tree_render()); - // Reset runtime counters so we measure hits from the warm run onwards. + // Reset runtime counters so we measure hits from the prefetch run onwards. cache.storage().stats(); let first_liquid_run = liquid_ctx.sql(sql).await.unwrap().collect().await.unwrap(); @@ -477,17 +526,23 @@ async fn test_provide_schema2() { } } - #[cfg(target_arch = "x86_64")] + // FSST breaks equal-gain symbol ties using target-specific HashMap iteration order. + // Canonicalize the known AArch64 totals to x86_64 while leaving unexpected totals visible. + #[cfg(target_arch = "aarch64")] + let snapshot = snapshot + .replace("usage.memory_bytes: 999980", "usage.memory_bytes: 1000915") + .replace("usage.memory_bytes: 1035369", "usage.memory_bytes: 1036304"); + + #[cfg(any(target_arch = "x86_64", target_arch = "aarch64"))] insta::assert_snapshot!(snapshot); - // Off x86_64 the byte-exact snapshot cannot match, because FSST picks a - // different symbol table (see above). Bound the figures instead of skipping - // the test: arm64 is a production target, so it still deserves a tripwire on - // a gross accounting regression. The plan text is only checked byte-for-byte - // on x86_64, but what this test covers besides it — the DataFusion-vs-liquid - // column equality, the cache hits, the tier split, the `Utf8`-declared schema - // over a `string_view` file — is architecture-independent and worth running. - #[cfg(not(target_arch = "x86_64"))] + // On any other target the byte-exact snapshot cannot match, because FSST + // picks a different symbol table (see above) and only the x86_64/aarch64 + // totals are known. Bound the figures instead of skipping the test: what this + // test covers besides the plan text — the DataFusion-vs-liquid column + // equality, the cache hits, the tier split, the `Utf8`-declared schema over a + // `string_view` file — is architecture-independent and worth running. + #[cfg(not(any(target_arch = "x86_64", target_arch = "aarch64")))] assert_memory_bytes_within_1pct( &snapshot, include_str!("snapshots/liquid_cache_datafusion_local__tests__provide_schema2.snap"), @@ -504,7 +559,7 @@ async fn test_provide_schema2() { /// The known architecture difference is ~0.1% (935 bytes in ~1 MiB), so 1% has an /// order of magnitude of headroom while still catching the kind of regression that /// matters — a buffer counted twice, or a tier accounted at the wrong size. -#[cfg(not(target_arch = "x86_64"))] +#[cfg(not(any(target_arch = "x86_64", target_arch = "aarch64")))] fn assert_memory_bytes_within_1pct(snapshot: &str, recorded: &str) { let actual = memory_bytes(snapshot); let expected = memory_bytes(recorded); @@ -539,7 +594,7 @@ fn assert_memory_bytes_within_1pct(snapshot: &str, recorded: &str) { /// /// Works on both the live snapshot and a committed `.snap` file: insta writes the /// snapshot body unindented after its YAML header, so the same prefix matches. -#[cfg(not(target_arch = "x86_64"))] +#[cfg(not(any(target_arch = "x86_64", target_arch = "aarch64")))] fn memory_bytes(snapshot: &str) -> Vec { snapshot .lines() @@ -614,8 +669,8 @@ async fn test_provide_schema_with_filter() { /// Covers the multi-partition scan path against the shared cache. /// -/// The tests above pin `repartition_file_min_size` above the test file size so -/// the scan stays single-partition and their traces stay reproducible. That pin +/// The tests above disable file-scan repartitioning outright so +/// the scan stays single-partition and their traces stay reproducible. That is /// is deliberate, but it also means nothing else exercises several scan /// partitions admitting into one cache concurrently — which is exactly what a /// default DataFusion 55 deployment does for any file over 1 MiB, since DF 55 @@ -703,3 +758,53 @@ async fn test_multi_partition_scan_shares_cache() { let (single_ctx, _single_cache) = build_ctx(single_config, single_dir.path()).await; assert_eq!(sorted_rows(&single_ctx, sql).await, second_run); } + +#[tokio::test] +async fn test_repartitioned_file_scan_cache_correctness() { + let reference_cache_dir = TempDir::new().unwrap(); + let parallel_cache_dir = TempDir::new().unwrap(); + let sql = r#"select "WatchID", "OS", "EventTime" from hits where "OS" <> 2 order by "WatchID" desc limit 10"#; + + let reference = run_sql_with_cache( + sql, + Box::new(TranscodeSqueezeEvict), + 1024 * 1024, + reference_cache_dir.path(), + ) + .await + .values; + + // DataFusion 55 lowered repartition_file_min_size from 10 MiB to 1 MiB, + // which splits the 2.3 MiB fixture into four concurrent scan partitions. + let mut config = SessionConfig::new(); + config.options_mut().execution.target_partitions = 4; + let (ctx, cache) = LiquidCacheLocalBuilder::new() + .with_max_memory_bytes(1024 * 1024) + .with_cache_dir(parallel_cache_dir.path().to_path_buf()) + .with_squeeze_policy(Box::new(TranscodeSqueezeEvict)) + .with_cache_policy(Box::new(LiquidPolicy::new())) + .build(config) + .await + .unwrap(); + ctx.register_parquet("hits", TEST_FILE, ParquetReadOptions::default()) + .await + .unwrap(); + + let plan = get_physical_plan(sql, &ctx).await; + let plan = format!( + "{}", + DisplayableExecutionPlan::new(plan.as_ref()).tree_render() + ); + assert!( + plan.contains("files: 4"), + "expected a repartitioned scan:\n{plan}" + ); + + assert_eq!(get_result(&ctx, sql).await, reference); + let entries_after_first_run = cache.storage().stats().total_entries; + assert_eq!(get_result(&ctx, sql).await, reference); + + let stats = cache.storage().stats(); + assert!(stats.runtime.get_with_selection > 0); + assert!(stats.total_entries >= entries_after_first_run); +} diff --git a/src/datafusion-local/src/tests/nested_filter.rs b/src/datafusion-local/src/tests/nested_filter.rs new file mode 100644 index 000000000..fb51344dc --- /dev/null +++ b/src/datafusion-local/src/tests/nested_filter.rs @@ -0,0 +1,98 @@ +use std::{fs::File, sync::Arc}; + +use arrow::{ + array::{AsArray, Int32Array, StructArray}, + datatypes::{DataType, Field, Fields, Schema}, + record_batch::RecordBatch, +}; +use datafusion::prelude::{ParquetReadOptions, SessionConfig}; +use parquet::arrow::ArrowWriter; +use tempfile::TempDir; + +use crate::LiquidCacheLocalBuilder; + +fn write_people(path: &std::path::Path) { + let person_fields = Fields::from(vec![Field::new("age", DataType::Int32, false)]); + let schema = Arc::new(Schema::new(vec![ + Field::new("person", DataType::Struct(person_fields.clone()), false), + Field::new("cohort", DataType::Int32, false), + ])); + let person = StructArray::new( + person_fields, + vec![Arc::new(Int32Array::from(vec![1, 1, 2, 2]))], + None, + ); + let batch = RecordBatch::try_new( + Arc::clone(&schema), + vec![ + Arc::new(person), + Arc::new(Int32Array::from(vec![10, 20, 20, 30])), + ], + ) + .unwrap(); + + let mut writer = ArrowWriter::try_new(File::create(path).unwrap(), schema, None).unwrap(); + writer.write(&batch).unwrap(); + writer.close().unwrap(); +} + +async fn query_rows(sql: &str) -> Vec<(i32, i32)> { + let temp_dir = TempDir::new().unwrap(); + let parquet_path = temp_dir.path().join("people.parquet"); + write_people(&parquet_path); + + let (ctx, _) = LiquidCacheLocalBuilder::new() + .with_cache_dir(temp_dir.path().to_path_buf()) + .build(SessionConfig::new()) + .await + .unwrap(); + ctx.register_parquet( + "people", + parquet_path.to_str().unwrap(), + ParquetReadOptions::default(), + ) + .await + .unwrap(); + + ctx.sql(sql) + .await + .unwrap() + .collect() + .await + .unwrap() + .into_iter() + .flat_map(|batch| { + let ages = batch + .column(0) + .as_primitive::(); + let cohorts = batch + .column(1) + .as_primitive::(); + (0..batch.num_rows()) + .map(|row| (ages.value(row), cohorts.value(row))) + .collect::>() + }) + .collect() +} + +#[tokio::test] +async fn nested_struct_field_filter_keeps_only_matching_rows() { + let rows = query_rows( + "SELECT person['age'], cohort \ + FROM people WHERE person['age'] = 2 ORDER BY cohort", + ) + .await; + + assert_eq!(rows, vec![(2, 20), (2, 30)]); +} + +#[tokio::test] +async fn nested_and_primitive_filters_are_both_applied() { + let rows = query_rows( + "SELECT person['age'], cohort \ + FROM people WHERE person['age'] = 2 AND cohort = 20", + ) + .await; + + assert_eq!(rows, vec![(2, 20)]); +} diff --git a/src/datafusion-local/src/tests/snapshots/liquid_cache_datafusion_local__tests__provide_schema2.snap b/src/datafusion-local/src/tests/snapshots/liquid_cache_datafusion_local__tests__provide_schema2.snap index 0a535fb1a..c127d3fe1 100644 --- a/src/datafusion-local/src/tests/snapshots/liquid_cache_datafusion_local__tests__provide_schema2.snap +++ b/src/datafusion-local/src/tests/snapshots/liquid_cache_datafusion_local__tests__provide_schema2.snap @@ -1,5 +1,6 @@ --- source: src/datafusion-local/src/tests/mod.rs +assertion_line: 428 expression: snapshot --- query[0]: SELECT * from default where log like '%hhj%' order by _timestamp @@ -190,7 +191,7 @@ RuntimeStatsSnapshot: eval_predicate: 0 get_squeezed_success: 0 get_squeezed_needs_io: 0 - try_read_liquid_calls: 0 + try_read_liquid_calls: 3 hit_date32_expression_calls: 0 read_io_count: 0 write_io_count: 0 diff --git a/src/datafusion-local/src/tests/squeeze.rs b/src/datafusion-local/src/tests/squeeze.rs index e32d21a38..b9863ae00 100644 --- a/src/datafusion-local/src/tests/squeeze.rs +++ b/src/datafusion-local/src/tests/squeeze.rs @@ -1,17 +1,24 @@ use arrow::{array::AsArray, datatypes::Int64Type, util::pretty::pretty_format_batches}; use tempfile::TempDir; +use datafusion::prelude::SessionConfig; + use crate::LiquidCacheLocalBuilder; const TEST_FILE: &str = "../../examples/nano_hits.parquet"; +fn squeeze_test_config() -> SessionConfig { + SessionConfig::new().with_repartition_file_scans(false) +} + #[tokio::test] async fn basic_squeeze() { let cache_dir = TempDir::new().unwrap(); let (ctx, cache) = LiquidCacheLocalBuilder::new() + .with_prefetch(false) .with_max_memory_bytes(1024 * 128) .with_cache_dir(cache_dir.path().to_path_buf()) - .build(super::cache_test_config()) + .build(squeeze_test_config()) .await .unwrap(); ctx.register_parquet("hits", TEST_FILE, Default::default()) @@ -36,9 +43,10 @@ async fn basic_squeeze() { async fn squeeze_strings() { let cache_dir = TempDir::new().unwrap(); let (ctx, cache) = LiquidCacheLocalBuilder::new() + .with_prefetch(false) .with_max_memory_bytes(1024 * 1024) .with_cache_dir(cache_dir.path().to_path_buf()) - .build(super::cache_test_config()) + .build(squeeze_test_config()) .await .unwrap(); ctx.register_parquet("hits", TEST_FILE, Default::default()) @@ -63,9 +71,10 @@ async fn squeeze_strings() { async fn squeeze_substrings_search() { let cache_dir = TempDir::new().unwrap(); let (ctx, cache) = LiquidCacheLocalBuilder::new() + .with_prefetch(false) .with_max_memory_bytes(1024 * 256) .with_cache_dir(cache_dir.path().to_path_buf()) - .build(super::cache_test_config()) + .build(squeeze_test_config()) .await .unwrap(); ctx.register_parquet("hits", TEST_FILE, Default::default()) @@ -87,9 +96,10 @@ async fn squeeze_substrings_search() { async fn squeeze_substrings_search_title() { let cache_dir = TempDir::new().unwrap(); let (ctx, cache) = LiquidCacheLocalBuilder::new() + .with_prefetch(false) .with_max_memory_bytes(1024 * 1024 * 4) .with_cache_dir(cache_dir.path().to_path_buf()) - .build(super::cache_test_config()) + .build(squeeze_test_config()) .await .unwrap(); ctx.register_parquet("hits", TEST_FILE, Default::default()) @@ -112,9 +122,10 @@ async fn squeeze_substrings_search_title() { async fn squeeze_distinct_search_phase() { let cache_dir = TempDir::new().unwrap(); let (ctx, cache) = LiquidCacheLocalBuilder::new() + .with_prefetch(false) .with_max_memory_bytes(1024 * 256) .with_cache_dir(cache_dir.path().to_path_buf()) - .build(super::cache_test_config()) + .build(squeeze_test_config()) .await .unwrap(); ctx.register_parquet("hits", TEST_FILE, Default::default()) diff --git a/src/datafusion-local/src/tests/unevaluable_conjunct.rs b/src/datafusion-local/src/tests/unevaluable_conjunct.rs index 0dfda2c16..0d7066917 100644 --- a/src/datafusion-local/src/tests/unevaluable_conjunct.rs +++ b/src/datafusion-local/src/tests/unevaluable_conjunct.rs @@ -3,11 +3,13 @@ //! //! `build_row_filter` splits the pushed-down predicate into conjuncts and builds //! one `FilterCandidate` per conjunct. `PushdownChecker` refuses a conjunct that -//! touches a nested column, and one that references a column absent from the -//! file schema. Those refusals used to be dropped silently, leaving the scan -//! applying a *strictly weaker* filter than the query asked for — and since -//! DataFusion removes the `FilterExec` when it pushes a predicate down, nothing -//! re-applies the dropped conjunct. +//! references a column absent from the table schema. Such a refusal used to be +//! dropped silently, leaving the scan applying a *strictly weaker* filter than +//! the query asked for — and since DataFusion removes the `FilterExec` when it +//! pushes a predicate down, nothing re-applies the dropped conjunct. +//! +//! A nested column is not one of these: it pushes down like any other, so a +//! struct in the predicate costs the scan nothing. //! //! The scan is now declined at plan time instead, so the predicate stays with //! the reader that planned it. @@ -48,13 +50,22 @@ fn write_t(path: &Path) { } async fn liquid_ctx(cache_dir: &Path) -> SessionContext { + liquid_ctx_with_cache(cache_dir).await.0 +} + +async fn liquid_ctx_with_cache( + cache_dir: &Path, +) -> ( + SessionContext, + liquid_cache_datafusion::LiquidCacheParquetRef, +) { std::fs::create_dir_all(cache_dir).unwrap(); - let (ctx, _cache) = LiquidCacheLocalBuilder::new() + let (ctx, cache) = LiquidCacheLocalBuilder::new() .with_cache_dir(cache_dir.to_path_buf()) .build(SessionConfig::new()) .await .unwrap(); - ctx + (ctx, cache) } async fn ids(ctx: &SessionContext, sql: &str) -> Vec { @@ -100,16 +111,13 @@ async fn nested_column_conjunct_is_still_applied() { let sql = "SELECT id FROM t WHERE id >= 0 AND st.a = 3"; - // The scan is declined rather than taken over with a filter that cannot - // evaluate `st.a`, so the plan still carries the whole predicate. + // A nested conjunct is evaluable now, so the scan is taken over and keeps the + // cache. What must not change is the answer: the conjunct is applied, not + // dropped. let plan = plan_of(&ctx, sql).await; assert!( - plan.contains("id@0 >= 0 AND get_field(st@1, a) = 3"), - "the unevaluable conjunct left the plan:\n{plan}" - ); - assert!( - !plan.contains("liquid_parquet"), - "the scan was taken over despite an unevaluable conjunct:\n{plan}" + plan.contains("liquid_parquet"), + "a nested conjunct should no longer cost the scan its cache:\n{plan}" ); // The first pass reads through the parquet fallback and would fill the cache; @@ -144,12 +152,14 @@ async fn sole_unevaluable_conjunct_is_still_applied() { } } -/// Declining a scan costs it the cache, so the refusal has to stay narrow: it is -/// a nested column *in the predicate* that the row filter cannot evaluate, not -/// the mere presence of one in the table or in the projection. Projecting `st.a` -/// is still cached, and so is a scan of a table that merely has a struct column. +/// A struct column costs the scan nothing, wherever it appears. The row filter +/// used to refuse any nested column outright — a bar inherited from DataFusion's +/// own `row_filter.rs` — which declined the whole scan to `ParquetSource` and so +/// lost the cache for every filtered query on a table carrying one. A column the +/// cache cannot transcode is simply held as Arrow and the predicate evaluates +/// against that, so nothing here needs declining. #[tokio::test] -async fn only_a_nested_predicate_costs_the_cache() { +async fn a_nested_column_never_costs_the_cache() { let dir = TempDir::new().unwrap(); let parquet = dir.path().join("t.parquet"); write_t(&parquet); @@ -166,26 +176,64 @@ async fn only_a_nested_predicate_costs_the_cache() { "SELECT st.a FROM t WHERE id >= 0", "SELECT id FROM t WHERE id >= 0", "SELECT id FROM t", + "SELECT id FROM t WHERE st.a = 3", + "SELECT id FROM t WHERE id >= 0 AND st.a = 3", ] { let plan = plan_of(&ctx, sql).await; assert!( plan.contains("liquid_parquet"), - "`{sql}` lost the cache; the refusal has widened:\n{plan}" + "`{sql}` lost the cache:\n{plan}" ); } + // And the nested predicates still return the right rows through the cache. for sql in [ "SELECT id FROM t WHERE st.a = 3", "SELECT id FROM t WHERE id >= 0 AND st.a = 3", ] { - let plan = plan_of(&ctx, sql).await; - assert!( - !plan.contains("liquid_parquet"), - "`{sql}` kept a filter it cannot evaluate:\n{plan}" - ); + for pass in ["cold", "warm"] { + assert_eq!(ids(&ctx, sql).await, vec![3], "{sql} ({pass})"); + } } } +/// The cache is not merely *kept* on a nested predicate, it is *used*: the struct +/// column is admitted and the warm pass reads it back from the cache. +/// +/// What such a scan does not get is the encoded-data fast path +/// (`eval_predicate`), because a struct root does not resolve to one cache column +/// id — see `convert_parquet_scan`'s docs. Pinned here so that closing that gap +/// shows up as a deliberate change to this test rather than passing unnoticed. +#[tokio::test] +async fn a_nested_predicate_is_served_from_the_cache() { + let dir = TempDir::new().unwrap(); + let parquet = dir.path().join("t.parquet"); + write_t(&parquet); + let (ctx, cache) = liquid_ctx_with_cache(&dir.path().join("cache")).await; + ctx.register_parquet( + "t", + parquet.to_str().unwrap(), + ParquetReadOptions::default(), + ) + .await + .unwrap(); + + let sql = "SELECT id FROM t WHERE st.a = 3"; + cache.storage().stats(); + assert_eq!(ids(&ctx, sql).await, vec![3], "cold"); + assert_eq!(ids(&ctx, sql).await, vec![3], "warm"); + + let stats = cache.storage().stats(); + assert!( + stats.total_entries > 0, + "the struct column was never admitted: {stats:?}" + ); + assert!( + stats.runtime.get_with_selection > 0 || stats.runtime.get > 0, + "the warm pass did not read the cache: {stats:?}" + ); +} + /// The other `PushdownChecker` refusal: a conjunct on a column that is not in the /// file schema. The table declares `extra`, the file does not have it, so every /// row's `extra` is NULL and `extra = 3` is never TRUE. diff --git a/src/datafusion-server/Cargo.toml b/src/datafusion-server/Cargo.toml index 20a0b3166..be6b79a65 100644 --- a/src/datafusion-server/Cargo.toml +++ b/src/datafusion-server/Cargo.toml @@ -2,6 +2,7 @@ name = "liquid-cache-datafusion-server" version = { workspace = true } edition = { workspace = true } +rust-version = { workspace = true } license = { workspace = true } readme = "README.md" description = { workspace = true } @@ -25,8 +26,8 @@ liquid-cache-common = { workspace = true } tempfile = { workspace = true } axum = "0.8.9" serde = { workspace = true } -tower-http = { version = "0.6.11", features = ["cors"] } -sysinfo = { version = "0.38.4", default-features = false, features = [ +tower-http = { version = "0.7.1", features = ["cors"] } +sysinfo = { version = "0.39.6", default-features = false, features = [ "component", "disk", "network", diff --git a/src/datafusion-server/src/admin_server/handlers.rs b/src/datafusion-server/src/admin_server/handlers.rs index 1be5b0be7..029f928aa 100644 --- a/src/datafusion-server/src/admin_server/handlers.rs +++ b/src/datafusion-server/src/admin_server/handlers.rs @@ -315,6 +315,9 @@ pub(crate) async fn start_flamegraph_handler( impl From<&Arc> for ExecutionPlanWithStats { fn from(plan: &Arc) -> Self { let metrics = plan.metrics().unwrap().aggregate_by_name(); + let statistics = StatisticsContext::new() + .compute(plan.as_ref(), &StatisticsArgs::new()) + .unwrap(); let mut metric_values = Vec::new(); for metric in metrics.iter() { metric_values.push(MetricValues { @@ -323,12 +326,8 @@ impl From<&Arc> for ExecutionPlanWithStats { }); } - let stats = StatisticsContext::new() - .compute(plan.as_ref(), &StatisticsArgs::new()) - .unwrap(); - let mut column_statistics = Vec::new(); - for (i, cs) in stats.column_statistics.iter().enumerate() { + for (i, cs) in statistics.column_statistics.iter().enumerate() { let min = if cs.min_value != Precision::Absent { Some(cs.min_value.to_string()) } else { @@ -376,8 +375,8 @@ impl From<&Arc> for ExecutionPlanWithStats { }) .collect(), statistics: Statistics { - num_rows: stats.num_rows.to_string(), - total_byte_size: stats.total_byte_size.to_string(), + num_rows: statistics.num_rows.to_string(), + total_byte_size: statistics.total_byte_size.to_string(), column_statistics, }, metrics: metric_values, diff --git a/src/datafusion/Cargo.toml b/src/datafusion/Cargo.toml index 0dfd8d590..c59d43ed1 100644 --- a/src/datafusion/Cargo.toml +++ b/src/datafusion/Cargo.toml @@ -2,6 +2,7 @@ name = "liquid-cache-datafusion" version = { workspace = true } edition = { workspace = true } +rust-version = { workspace = true } license = { workspace = true } readme = "README.md" description = { workspace = true } @@ -12,6 +13,7 @@ arrow = { workspace = true } arrow-schema = { workspace = true } parquet = { workspace = true } datafusion = { workspace = true } +datafusion-datasource = { workspace = true } futures = { workspace = true } tokio = { workspace = true } ahash = { workspace = true } diff --git a/src/datafusion/bench/filter_pushdown.rs b/src/datafusion/bench/filter_pushdown.rs index 33244c733..fa8f05b56 100644 --- a/src/datafusion/bench/filter_pushdown.rs +++ b/src/datafusion/bench/filter_pushdown.rs @@ -15,7 +15,7 @@ use datafusion::physical_expr::PhysicalExpr; use datafusion::physical_expr::expressions::{BinaryExpr, Literal}; use datafusion::physical_plan::expressions::Column; use datafusion::physical_plan::metrics; -use liquid_cache_datafusion::cache::{BatchID, LiquidCacheParquet}; +use liquid_cache_datafusion::cache::{BatchID, LiquidCacheParquet, ParquetFileIdentity}; use parquet::arrow::ArrowWriter; use parquet::arrow::arrow_reader::{ArrowReaderMetadata, ArrowReaderOptions}; use rand::RngExt as _; @@ -53,7 +53,13 @@ fn setup_cache() -> (Arc, tempfile::TempDir) { )); let field = Arc::new(Field::new("test_column", DataType::Int32, false)); let schema = Arc::new(Schema::new(vec![field.clone()])); - let file = cache.register_or_get_file("test_file.parquet".to_string(), schema); + let file = cache.register_or_get_file( + ParquetFileIdentity::new( + datafusion::execution::object_store::ObjectStoreUrl::local_filesystem(), + "test_file.parquet".to_string(), + ), + schema, + ); let row_group = file.create_row_group(0, vec![]); (row_group.get_column(0).unwrap(), tmp_dir) } diff --git a/src/datafusion/src/cache/column.rs b/src/datafusion/src/cache/column.rs index 1f582fd86..fc02ccebc 100644 --- a/src/datafusion/src/cache/column.rs +++ b/src/datafusion/src/cache/column.rs @@ -5,12 +5,14 @@ use arrow::{ record_batch::RecordBatch, }; use arrow_schema::{ArrowError, Field, Schema}; -use liquid_cache::cache::{CacheExpression, CacheFull, LiquidCache, LiquidExpr}; +use liquid_cache::cache::{ + CacheEntry, CacheExpression, CacheFull, LiquidCache, LiquidExpr, PrefetchResult, +}; use parquet::arrow::arrow_reader::ArrowPredicate; use crate::{ LiquidPredicate, - cache::{BatchID, ColumnAccessPath, ParquetArrayID}, + cache::{BatchID, ColumnAccessPath, ParquetArrayID, RowGroupSnapshots}, }; use std::sync::Arc; @@ -21,6 +23,7 @@ pub struct CachedColumn { field: Arc, column_path: ColumnAccessPath, expression: Option>, + snapshots: Arc, } /// A reference to a cached column. @@ -35,6 +38,13 @@ pub enum InsertArrowArrayError { CacheFull, } +pub(crate) enum PrefetchOutcome { + Snapshotted, + AlreadySnapshotted, + Squeezed, + Missing, +} + impl From for InsertArrowArrayError { fn from(_: CacheFull) -> Self { Self::CacheFull @@ -48,6 +58,7 @@ impl CachedColumn { column_access_path: ColumnAccessPath, expression: Option>, is_predicate_column: bool, + snapshots: Arc, ) -> Self { // Register the column's squeeze hint. Squeeze hints are column-scoped; // `ParquetCacheMetadata` keys them by column (the batch id is masked @@ -70,6 +81,7 @@ impl CachedColumn { cache_store, column_path: column_access_path, expression, + snapshots, } } @@ -78,8 +90,12 @@ impl CachedColumn { self.column_path.entry_id(batch_id) } - pub(crate) fn is_cached(&self, batch_id: BatchID) -> bool { - self.cache_store.is_cached(&self.entry_id(batch_id).into()) + pub(crate) fn contains(&self, batch_id: BatchID) -> bool { + self.cache_store.contains(&self.entry_id(batch_id).into()) + } + + pub(crate) fn snapshot_entry(&self, batch_id: BatchID) -> Option> { + self.snapshots.get(&self.entry_id(batch_id).into()) } /// Returns the Arrow field metadata for this cached column. @@ -112,11 +128,24 @@ impl CachedColumn { ); if let Some(liquid_expr) = liquid_expr - && let Some(boolean_array) = self - .cache_store - .eval_predicate(&entry_id, &liquid_expr) - .with_selection(filter) - .await + && let Some(boolean_array) = match self.snapshots.get(&entry_id) { + Some(entry) => { + self.cache_store + .eval_predicate_on_entry( + &entry_id, + entry.as_ref(), + Some(filter), + &liquid_expr, + ) + .await + } + None => { + self.cache_store + .eval_predicate(&entry_id, &liquid_expr) + .with_selection(filter) + .await + } + } { let predicate_filter = match boolean_array.null_count() { 0 => boolean_array, @@ -159,6 +188,17 @@ impl CachedColumn { filter: &BooleanBuffer, ) -> Option { let entry_id = self.entry_id(batch_id).into(); + if let Some(entry) = self.snapshots.get(&entry_id) { + return self + .cache_store + .read_entry( + &entry_id, + entry.as_ref(), + Some(filter), + self.expression.as_deref(), + ) + .await; + } self.cache_store .get(&entry_id) .with_selection(filter) @@ -179,7 +219,7 @@ impl CachedColumn { batch_id: BatchID, array: ArrayRef, ) -> Result<(), InsertArrowArrayError> { - if self.is_cached(batch_id) { + if self.contains(batch_id) { return Err(InsertArrowArrayError::AlreadyCached); } @@ -188,4 +228,26 @@ impl CachedColumn { .await?; Ok(()) } + + pub(crate) fn insert_snapshot(&self, batch_id: BatchID, array: ArrayRef) { + self.snapshots.insert( + self.entry_id(batch_id).into(), + Arc::new(CacheEntry::memory_arrow(array)), + ); + } + + pub(crate) async fn prefetch_snapshot(&self, batch_id: BatchID) -> PrefetchOutcome { + let entry_id = self.entry_id(batch_id).into(); + if self.snapshots.get(&entry_id).is_some() { + return PrefetchOutcome::AlreadySnapshotted; + } + match self.cache_store.prefetch(&entry_id).await { + PrefetchResult::Snapshot(entry) => { + self.snapshots.insert(entry_id, entry); + PrefetchOutcome::Snapshotted + } + PrefetchResult::Squeezed => PrefetchOutcome::Squeezed, + PrefetchResult::Absent => PrefetchOutcome::Missing, + } + } } diff --git a/src/datafusion/src/cache/mod.rs b/src/datafusion/src/cache/mod.rs index 814d0d324..320b99f08 100644 --- a/src/datafusion/src/cache/mod.rs +++ b/src/datafusion/src/cache/mod.rs @@ -3,14 +3,19 @@ use crate::io::ParquetCacheMetadata; use crate::reader::{LiquidPredicate, extract_multi_column_or}; -use crate::sync::Mutex; +use crate::sync::{Mutex, RwLock}; use ahash::AHashMap; use arrow::array::{BooleanArray, RecordBatch, RecordBatchOptions}; use arrow::buffer::BooleanBuffer; use arrow_schema::{ArrowError, Field, Schema, SchemaRef}; +use datafusion::common::tree_node::{Transformed, TreeNode}; +use datafusion::execution::object_store::ObjectStoreUrl; +use datafusion::physical_expr::PhysicalExpr; +use datafusion::physical_expr::expressions::Column; use liquid_cache::cache::squeeze_policies::SqueezePolicy; use liquid_cache::cache::{ - CacheExpression, CachePolicy, EventTrace, HydrationPolicy, LiquidCache, LiquidCacheBuilder, + CacheEntry, CacheExpression, CachePolicy, EntryID, EventTrace, HydrationPolicy, LiquidCache, + LiquidCacheBuilder, }; use parquet::arrow::arrow_reader::ArrowPredicate; use std::collections::HashMap; @@ -22,8 +27,8 @@ mod column; mod id; mod stats; -pub(crate) use column::InsertArrowArrayError; pub use column::{CachedColumn, CachedColumnRef}; +pub(crate) use column::{InsertArrowArrayError, PrefetchOutcome}; pub(crate) use id::ColumnAccessPath; pub use id::{BatchID, ParquetArrayID}; @@ -34,9 +39,53 @@ pub use id::{BatchID, ParquetArrayID}; /// [`LiquidParquetSource`](crate::LiquidParquetSource) that opens the file. pub type ColumnSqueezeHints = HashMap>; +/// The identity of a Parquet object within an object store. +/// +/// Object paths are only unique within their object store, so both components +/// are required to keep cached data from different stores isolated. +#[derive(Clone, Debug, Eq, Hash, PartialEq)] +pub struct ParquetFileIdentity { + object_store_url: ObjectStoreUrl, + path: String, +} + +impl ParquetFileIdentity { + /// Create an identity from an object store URL and an object path. + pub fn new(object_store_url: ObjectStoreUrl, path: String) -> Self { + Self { + object_store_url, + path, + } + } +} + /// One column of a row group: (file column index, field, squeeze hint, is-predicate). type CachedColumnSpec = (u64, Arc, Option>, bool); +#[derive(Default, Debug)] +pub(crate) struct RowGroupSnapshots { + entries: RwLock>>, + selections: RwLock>, +} + +impl RowGroupSnapshots { + pub(crate) fn get(&self, entry_id: &EntryID) -> Option> { + self.entries.read().unwrap().get(entry_id).cloned() + } + + pub(crate) fn insert(&self, entry_id: EntryID, entry: Arc) { + self.entries.write().unwrap().insert(entry_id, entry); + } + + pub(crate) fn selection(&self, batch_id: BatchID) -> Option { + self.selections.read().unwrap().get(&batch_id).cloned() + } + + pub(crate) fn insert_selection(&self, batch_id: BatchID, selection: BooleanBuffer) { + self.selections.write().unwrap().insert(batch_id, selection); + } +} + #[derive(Default, Debug)] struct ColumnMaps { // invariant: Arc::ptr_eq(map[field.name()], map[field.id()]) @@ -49,6 +98,7 @@ struct ColumnMaps { pub struct CachedRowGroup { columns: ColumnMaps, cache_store: Arc, + snapshots: Arc, } impl CachedRowGroup { @@ -60,6 +110,7 @@ impl CachedRowGroup { row_group_idx: u64, file_idx: u64, columns: &[CachedColumnSpec], + snapshots: Arc, ) -> Self { let mut column_maps = ColumnMaps::default(); for (column_id, field, expression, is_predicate_column) in columns { @@ -70,6 +121,7 @@ impl CachedRowGroup { column_access_path, expression.clone(), *is_predicate_column, + Arc::clone(&snapshots), )); column_maps.by_id.insert(*column_id, column.clone()); column_maps.by_name.insert(field.name().to_string(), column); @@ -78,6 +130,7 @@ impl CachedRowGroup { Self { columns: column_maps, cache_store, + snapshots, } } @@ -103,6 +156,10 @@ impl CachedRowGroup { self.columns.by_name.get(unqualified).cloned() } + pub(crate) fn snapshot_selection(&self, batch_id: BatchID) -> Option { + self.snapshots.selection(batch_id) + } + /// Evaluate a predicate on a row group. #[fastrace::trace] pub async fn evaluate_selection_with_predicate( @@ -129,7 +186,28 @@ impl CachedRowGroup { for (col_name, expr) in column_exprs { let column = self.get_column_by_name(col_name)?; - let liquid_expr = column.liquid_expr_for_predicate(Arc::clone(&expr)); + let snapshot_liquid = match column.snapshot_entry(batch_id) { + Some(entry) => match entry.as_ref() { + CacheEntry::MemoryLiquid(array) => Some(Arc::clone(array)), + _ => { + combined_buffer = None; + break; + } + }, + None => None, + }; + let expr = expr + .transform_up(|expr| { + if let Some(column) = expr.downcast_ref::() { + Ok(Transformed::yes(Arc::new(Column::new(column.name(), 0)) + as Arc)) + } else { + Ok(Transformed::no(expr)) + } + }) + .ok()? + .data; + let liquid_expr = column.liquid_expr_for_predicate(expr); let liquid_expr = match liquid_expr { Some(expr) => expr, None => { @@ -138,7 +216,10 @@ impl CachedRowGroup { } }; let entry_id = column.entry_id(batch_id).into(); - let liquid_array = self.cache_store.try_read_liquid(&entry_id).await; + let liquid_array = match snapshot_liquid { + Some(array) => Some(array), + None => self.cache_store.try_read_liquid(&entry_id).await, + }; let liquid_array = match liquid_array { None => { combined_buffer = None; @@ -215,6 +296,15 @@ impl CachedFile { &self, row_group_id: u64, predicate_column_ids: Vec, + ) -> CachedRowGroupRef { + self.create_row_group_with_snapshots(row_group_id, predicate_column_ids, Arc::default()) + } + + pub(crate) fn create_row_group_with_snapshots( + &self, + row_group_id: u64, + predicate_column_ids: Vec, + snapshots: Arc, ) -> CachedRowGroupRef { let columns: Vec = self .file_schema @@ -238,6 +328,7 @@ impl CachedFile { row_group_id, self.file_id, &columns, + snapshots, )) } @@ -258,8 +349,8 @@ pub(crate) type CachedFileRef = Arc; /// The main cache structure. #[derive(Debug)] pub struct LiquidCacheParquet { - /// Map file path to file id. - files: Mutex>, + /// Map object-store-qualified file identity to file id. + files: Mutex>, cache_store: Arc, @@ -331,23 +422,23 @@ impl LiquidCacheParquet { /// Register a file in the cache. pub fn register_or_get_file( &self, - file_path: String, + file_identity: ParquetFileIdentity, full_file_schema: SchemaRef, ) -> CachedFileRef { - self.register_or_get_file_with_hints(file_path, full_file_schema, Arc::default()) + self.register_or_get_file_with_hints(file_identity, full_file_schema, Arc::default()) } /// Register a file in the cache, attaching typed squeeze hints derived from /// the query plan (keyed by file-schema column name). pub fn register_or_get_file_with_hints( &self, - file_path: String, + file_identity: ParquetFileIdentity, full_file_schema: SchemaRef, squeeze_hints: Arc, ) -> CachedFileRef { let mut files = self.files.lock().unwrap(); let file_id = *files - .entry(file_path.clone()) + .entry(file_identity) .or_insert_with(|| self.current_file_id.fetch_add(1, Ordering::Relaxed)); drop(files); @@ -437,7 +528,7 @@ mod tests { use super::*; use crate::cache::{CachedRowGroupRef, LiquidCacheParquet}; use crate::reader::FilterCandidateBuilder; - use arrow::array::{Array, Int32Array}; + use arrow::array::{Array, ArrayRef, Int32Array, StringViewArray}; use arrow::buffer::BooleanBuffer; use arrow::datatypes::{DataType, Field, Schema}; use arrow::record_batch::RecordBatch; @@ -466,10 +557,160 @@ mod tests { Box::new(AlwaysHydrate::new()), ) .await; - let file = cache.register_or_get_file("test".to_string(), schema); + let file = cache.register_or_get_file( + ParquetFileIdentity::new(ObjectStoreUrl::local_filesystem(), "test".to_string()), + schema, + ); file.create_row_group(0, vec![]) } + async fn setup_liquid_cache( + batch_size: usize, + schema: SchemaRef, + max_memory_bytes: usize, + ) -> CachedRowGroupRef { + let tmp_dir = tempfile::tempdir().unwrap(); + let store = crate::test_utils::mount_test_store(tmp_dir.path()).await; + let cache = LiquidCacheParquet::new( + batch_size, + max_memory_bytes, + usize::MAX, + store, + Box::new(LiquidPolicy::new()), + Box::new(TranscodeSqueezeEvict), + Box::new(AlwaysHydrate::new()), + ) + .await; + cache + .register_or_get_file( + ParquetFileIdentity::new(ObjectStoreUrl::local_filesystem(), "test".to_string()), + schema, + ) + .create_row_group(0, vec![0, 1]) + } + + fn build_predicate( + schema: &SchemaRef, + arrays: Vec, + expr: Arc, + ) -> LiquidPredicate { + let tmp_meta = tempfile::NamedTempFile::new().unwrap(); + let mut writer = + ArrowWriter::try_new(tmp_meta.reopen().unwrap(), Arc::clone(schema), None).unwrap(); + writer + .write(&RecordBatch::try_new(Arc::clone(schema), arrays).unwrap()) + .unwrap(); + writer.close().unwrap(); + let reader = std::fs::File::open(tmp_meta.path()).unwrap(); + let metadata = ArrowReaderMetadata::load(&reader, ArrowReaderOptions::new()).unwrap(); + let candidate = FilterCandidateBuilder::new(expr, Arc::clone(schema)) + .build(metadata.metadata()) + .unwrap() + .unwrap(); + let projection = candidate.projection(metadata.metadata()); + LiquidPredicate::try_new(candidate, projection).unwrap() + } + + fn equals(name: &str, index: usize, value: ScalarValue) -> Arc { + Arc::new(BinaryExpr::new( + Arc::new(Column::new(name, index)), + Operator::Eq, + Arc::new(Literal::new(value)), + )) + } + + fn dummy(len: usize) -> ArrayRef { + Arc::new(Int32Array::from(vec![0; len])) + } + + async fn insert_liquid_columns( + row_group: &CachedRowGroupRef, + batch_id: BatchID, + arrays: [ArrayRef; 3], + ) { + for (column_id, array) in arrays.into_iter().enumerate() { + row_group + .get_column(column_id as u64) + .unwrap() + .insert(batch_id, array) + .await + .unwrap(); + } + assert_eq!(row_group.cache_store.stats().memory_liquid_entries, 2); + } + + #[tokio::test] + async fn or_fast_path_evaluates_on_liquid() { + let batch_size = 1024; + let schema = Arc::new(Schema::new(vec![ + Field::new("a", DataType::Int32, false), + Field::new("b", DataType::Int32, false), + Field::new("dummy", DataType::Int32, false), + ])); + let a: ArrayRef = Arc::new(Int32Array::from_iter_values( + (0..batch_size).map(|i| (i % 4) as i32 + 1), + )); + let b: ArrayRef = Arc::new(Int32Array::from_iter_values( + (0..batch_size).map(|i| (i % 5) as i32 * 10), + )); + let budget = a.get_array_memory_size() + b.get_array_memory_size(); + let row_group = setup_liquid_cache(batch_size, schema.clone(), budget).await; + let batch_id = BatchID::from_row_id(0, batch_size); + insert_liquid_columns(&row_group, batch_id, [a.clone(), b.clone(), dummy(1)]).await; + let expr = Arc::new(BinaryExpr::new( + equals("a", 0, ScalarValue::Int32(Some(3))), + Operator::Or, + equals("b", 1, ScalarValue::Int32(Some(20))), + )); + let mut predicate = build_predicate(&schema, vec![a, b, dummy(batch_size)], expr); + row_group.cache_store.stats(); + let selection = BooleanBuffer::new_set(batch_size); + let result = row_group + .evaluate_selection_with_predicate(batch_id, &selection, &mut predicate) + .await + .unwrap() + .unwrap(); + let expected = BooleanBuffer::collect_bool(batch_size, |i| i % 4 == 2 || i % 5 == 2); + assert_eq!(result, BooleanArray::new(expected, None)); + assert!(row_group.cache_store.stats().runtime.try_read_liquid_calls >= 2); + } + + #[tokio::test] + async fn or_fast_path_on_strings() { + let batch_size = 128; + let schema = Arc::new(Schema::new(vec![ + Field::new("city", DataType::Utf8View, false), + Field::new("name", DataType::Utf8View, false), + Field::new("dummy", DataType::Int32, false), + ])); + let city: ArrayRef = Arc::new(StringViewArray::from_iter_values( + (0..batch_size).map(|i| if i % 5 == 2 { "Tokyo" } else { "Paris" }), + )); + let name: ArrayRef = Arc::new(StringViewArray::from_iter_values( + (0..batch_size).map(|i| if i % 4 == 1 { "Bob" } else { "Alice" }), + )); + let budget = city.get_array_memory_size() + name.get_array_memory_size(); + let row_group = setup_liquid_cache(batch_size, schema.clone(), budget).await; + let batch_id = BatchID::from_row_id(0, batch_size); + insert_liquid_columns(&row_group, batch_id, [city.clone(), name.clone(), dummy(1)]).await; + let expr = Arc::new(BinaryExpr::new( + equals("name", 1, ScalarValue::Utf8View(Some("Bob".into()))), + Operator::Or, + equals("city", 0, ScalarValue::Utf8View(Some("Tokyo".into()))), + )); + let mut predicate = build_predicate(&schema, vec![city, name, dummy(batch_size)], expr); + row_group.cache_store.stats(); + let selection = BooleanBuffer::new_set(batch_size); + let result = row_group + .evaluate_selection_with_predicate(batch_id, &selection, &mut predicate) + .await + .unwrap() + .unwrap(); + let expected = BooleanBuffer::collect_bool(batch_size, |i| i % 4 == 1 || i % 5 == 2); + assert_eq!(result, BooleanArray::new(expected, None)); + assert!(row_group.cache_store.stats().runtime.try_read_liquid_calls >= 2); + } + /// Issue #19: `NOT (s = s)` simplifies to `s IS NULL AND NULL`, so a conjunct /// that reads no column reaches the row filter. It has to survive candidate /// building and then evaluate against the selection's row count — an diff --git a/src/datafusion/src/cache/stats.rs b/src/datafusion/src/cache/stats.rs index 474abcb2d..88fe12c05 100644 --- a/src/datafusion/src/cache/stats.rs +++ b/src/datafusion/src/cache/stats.rs @@ -164,7 +164,7 @@ impl LiquidCacheParquet { mod tests { use std::io::Read; - use crate::cache::id::BatchID; + use crate::cache::{ParquetFileIdentity, id::BatchID}; use super::*; use arrow::{ @@ -207,7 +207,13 @@ mod tests { let mut memory_size_sum = 0; for file_no in 0..8 { let file_name = format!("test_{file_no}.parquet"); - let file = cache.register_or_get_file(file_name, schema.clone()); + let file = cache.register_or_get_file( + ParquetFileIdentity::new( + datafusion::execution::object_store::ObjectStoreUrl::local_filesystem(), + file_name, + ), + schema.clone(), + ); for rg in 0..8 { let row_group = file.create_row_group(rg, vec![]); for col in 0..8 { diff --git a/src/datafusion/src/optimizers/mod.rs b/src/datafusion/src/optimizers/mod.rs index 3bb67d2d7..bc36551bd 100644 --- a/src/datafusion/src/optimizers/mod.rs +++ b/src/datafusion/src/optimizers/mod.rs @@ -91,6 +91,7 @@ pub struct LocalModeOptimizer { /// left as a vanilla parquet read instead of being wrapped by LiquidCache. /// `None` means cache every scan. admission: Option, + prefetch: bool, } impl LocalModeOptimizer { @@ -99,15 +100,19 @@ impl LocalModeOptimizer { Self { cache, admission: None, + prefetch: true, } } /// Create an optimizer with an existing cache instance pub fn with_cache(cache: LiquidCacheParquetRef) -> Self { - Self { - cache, - admission: None, - } + Self::new(cache) + } + + /// Enable or disable row-group prefetching. + pub fn with_prefetch(mut self, prefetch: bool) -> Self { + self.prefetch = prefetch; + self } /// Enable the footprint-based admission gate. A parquet scan is cached only @@ -155,6 +160,7 @@ impl PhysicalOptimizerRule for LocalModeOptimizer { let analysis = HintAnalyzer::analyze(&plan); let cache = self.cache.clone(); let admission = self.admission; + let prefetch = self.prefetch; // The gate sizes against both cache tiers, not just RAM: when a scan // overflows memory its entries spill to the on-disk liquid tier (NVMe) // instead of thrashing. They are weighted differently (see @@ -173,7 +179,7 @@ impl PhysicalOptimizerRule for LocalModeOptimizer { { return None; } - convert_parquet_scan(node, &cache, hints) + convert_parquet_scan(node, &cache, hints, prefetch) }; Ok(squeeze_hint::rewrite_with_hints( plan, @@ -203,7 +209,7 @@ pub fn rewrite_data_source_plan_with_hints( hints: &ColumnSqueezeHints, ) -> Arc { plan.transform_up( - |node| match convert_parquet_scan(&node, cache, hints.clone()) { + |node| match convert_parquet_scan(&node, cache, hints.clone(), true) { Some(new_node) => Ok(Transformed::new( new_node, true, @@ -627,12 +633,12 @@ fn unproducible_virtual_columns( /// this rule runs, DataFusion has already removed the `FilterExec` on the /// strength of `ParquetSource` accepting the whole predicate (both entry points /// force `execution.parquet.pushdown_filters`, so that removal always happens), -/// and the scan is the only place the predicate is applied. The liquid row filter -/// is stricter than DataFusion's own — it refuses nested columns, which upstream -/// handles — and it used to drop what it could not evaluate, running the scan -/// with a strictly weaker filter than the query asked for (issues #21, #23). -/// Declining the scan hands the predicate back to the reader that planned it, -/// which applies all of it. +/// and the scan is the only place the predicate is applied. What the liquid row +/// filter cannot evaluate is a reference to a column that is in no schema the +/// scan can read; it used to drop such a conjunct, running the scan with a +/// strictly weaker filter than the query asked for (issues #21, #23). Declining +/// the scan hands the predicate back to the reader that planned it, which +/// applies all of it. /// /// Two things this is not, and why: /// @@ -645,16 +651,21 @@ fn unproducible_virtual_columns( /// conversion from gating would fix local mode — but not the server, which /// receives a fragment whose `FilterExec` the *client* already removed. The /// server has no pushdown negotiation to join, so this gate is needed either way. -/// - **Not teaching the row filter about nested columns.** Upstream evaluates -/// `st.a = 3` by building a leaf-level `ProjectionMask` from struct field paths. -/// Liquid masks by root (`ProjectionMask::roots`) and `get_predicate_column_id` -/// reads the mask's leaf bits back as cache column ids, so a struct root — one -/// column, several leaves — would address several cache entries. Lifting that -/// means changing the cache's column-id model, well past a correctness fix. +/// - **Not a nested column.** A struct in the predicate no longer reaches this +/// bypass at all: `PushdownChecker` stopped refusing nested columns, so such a +/// scan converts, caches its columns and reads them back from the cache like +/// any other. What it does *not* get is the encoded-data fast path — liquid +/// masks by root (`ProjectionMask::roots`) while `get_predicate_column_id` +/// reads the mask's leaf bits back as cache column ids, so a struct root (one +/// column, several leaves) does not resolve to a single cache entry, and the +/// predicate falls through to evaluating materialised Arrow. Closing that gap +/// means changing the cache's column-id model, which is why it is still open; +/// it costs speed on struct predicates, not correctness or cache residency. fn convert_parquet_scan( node: &Arc, cache: &LiquidCacheParquetRef, hints: ColumnSqueezeHints, + prefetch: bool, ) -> Option> { let data_source_exec = node.downcast_ref::()?; let (file_scan_config, parquet_source) = @@ -699,7 +710,8 @@ fn convert_parquet_scan( let new_source = LiquidParquetSource::from_parquet_source(parquet_source.clone(), cache.clone()) - .with_squeeze_hints(Arc::new(hints)); + .with_squeeze_hints(Arc::new(hints)) + .with_prefetch(prefetch); let mut new_config = file_scan_config.clone(); new_config.file_source = Arc::new(new_source); @@ -709,11 +721,25 @@ fn convert_parquet_scan( #[cfg(test)] mod tests { - use datafusion::{datasource::physical_plan::FileScanConfig, prelude::SessionContext}; + use std::{fs::File, path::Path}; + + use arrow::{array::Int32Array, record_batch::RecordBatch}; + use arrow_schema::{DataType, Field, Schema}; + use datafusion::{ + common::{ScalarValue, stats::Precision}, + datasource::physical_plan::{FileScanConfig, FileSource}, + logical_expr::Operator, + physical_expr::expressions::{BinaryExpr, Column, Literal}, + physical_plan::{ + PhysicalExpr, collect, display::DisplayableExecutionPlan, filter_pushdown::PushedDown, + }, + prelude::SessionContext, + }; use liquid_cache::{ cache::{AlwaysHydrate, squeeze_policies::TranscodeSqueezeEvict}, cache_policies::LiquidPolicy, }; + use parquet::{arrow::ArrowWriter, file::properties::WriterProperties}; use crate::LiquidCacheParquet; @@ -812,7 +838,7 @@ mod tests { // Positive control: an ordinary scan is still handed to the cache. let plain = scan(TableSchema::builder(Arc::clone(&file_schema)).build()); assert!( - convert_parquet_scan(&plain, &cache, ColumnSqueezeHints::default()).is_some(), + convert_parquet_scan(&plain, &cache, ColumnSqueezeHints::default(), true).is_some(), "an ordinary scan must still convert to the liquid source" ); @@ -828,16 +854,18 @@ mod tests { .build(), ); assert!( - convert_parquet_scan(&positional, &cache, ColumnSqueezeHints::default()).is_none(), + convert_parquet_scan(&positional, &cache, ColumnSqueezeHints::default(), true) + .is_none(), "a scan reading a virtual column must stay on ParquetSource" ); } - async fn rewrite_plan_inner(plan: Arc) { - let expected_schema = plan.schema(); - let tmp_dir = tempfile::tempdir().unwrap(); - let store = crate::test_utils::mount_test_store(tmp_dir.path()).await; - let liquid_cache = Arc::new( + async fn make_cache(path: &Path) -> LiquidCacheParquetRef { + // Through the fork's mount helper, not `t4::mount`: it is the one + // place that picks the store's I/O mode, and requesting DIRECT I/O + // outright fails off Linux. `clippy.toml` bans the direct call. + let store = crate::test_utils::mount_test_store(path).await; + Arc::new( LiquidCacheParquet::new( 8192, 1000000, @@ -848,7 +876,32 @@ mod tests { Box::new(AlwaysHydrate::new()), ) .await, - ); + ) + } + + fn liquid_source(plan: &Arc) -> LiquidParquetSource { + let mut source = None; + plan.apply(|node| { + if let Some(plan) = node.downcast_ref::() { + let config = plan.data_source().downcast_ref::().unwrap(); + source = Some( + config + .file_source() + .downcast_ref::() + .unwrap() + .clone(), + ); + } + Ok(TreeNodeRecursion::Continue) + }) + .unwrap(); + source.unwrap() + } + + async fn rewrite_plan_inner(plan: Arc) -> Arc { + let expected_schema = plan.schema(); + let tmp_dir = tempfile::tempdir().unwrap(); + let liquid_cache = make_cache(tmp_dir.path()).await; let rewritten = rewrite_data_source_plan(plan, &liquid_cache); rewritten @@ -865,6 +918,8 @@ mod tests { Ok(TreeNodeRecursion::Continue) }) .unwrap(); + + rewritten } /// Regression: a `get_field` on a struct column is pushed into the scan @@ -970,7 +1025,116 @@ mod tests { .await .unwrap(); let plan = df.create_physical_plan().await.unwrap(); - rewrite_plan_inner(plan.clone()).await; + let rewritten = rewrite_plan_inner(plan).await; + + let displayed = DisplayableExecutionPlan::new(rewritten.as_ref()) + .indent(true) + .to_string(); + assert!(displayed.contains("predicate="), "{displayed}"); + + rewritten + .apply(|node| { + if let Some(plan) = node.downcast_ref::() { + let statistics = plan.data_source().partition_statistics(None)?; + assert!(!matches!(statistics.num_rows, Precision::Exact(_))); + } + Ok(TreeNodeRecursion::Continue) + }) + .unwrap(); + + // Supported filters are conjoined onto the predicate; unsupported ones + // are handed back to the parent. + let source = liquid_source(&rewritten); + let url_index = source.table_schema().file_schema().index_of("URL").unwrap(); + let supported: Arc = Arc::new(BinaryExpr::new( + Arc::new(Column::new("URL", url_index)), + Operator::Eq, + Arc::new(Literal::new(ScalarValue::Utf8(Some( + "https://example.com".into(), + )))), + )); + let unsupported: Arc = Arc::new(BinaryExpr::new( + Arc::new(Column::new("missing", 0)), + Operator::Eq, + Arc::new(Literal::new(ScalarValue::Utf8(Some("value".into())))), + )); + let result = source + .try_pushdown_filters( + vec![supported, unsupported], + &datafusion::config::ConfigOptions::new(), + ) + .unwrap(); + assert!(matches!( + result.filters.as_slice(), + [PushedDown::Yes, PushedDown::No] + )); + let predicate = result.updated_node.unwrap().filter().unwrap().to_string(); + assert!(predicate.contains(" AND "), "{predicate}"); + assert!(predicate.contains("https://example.com"), "{predicate}"); + assert!(!predicate.contains("missing"), "{predicate}"); + } + + fn write_bloom_file(path: &Path) { + let schema = Arc::new(Schema::new(vec![Field::new("a", DataType::Int32, false)])); + let properties = WriterProperties::builder() + .set_bloom_filter_enabled(true) + .build(); + let mut writer = ArrowWriter::try_new( + File::create(path).unwrap(), + schema.clone(), + Some(properties), + ) + .unwrap(); + for values in [[1, 2, 4], [1, 3, 4]] { + writer + .write( + &RecordBatch::try_new( + schema.clone(), + vec![Arc::new(Int32Array::from(values.to_vec()))], + ) + .unwrap(), + ) + .unwrap(); + writer.flush().unwrap(); + } + writer.close().unwrap(); + } + + #[tokio::test] + async fn prunes_row_group_with_bloom_filter() { + let tmp_dir = tempfile::tempdir().unwrap(); + let parquet_path = tmp_dir.path().join("bloom.parquet"); + write_bloom_file(&parquet_path); + + let ctx = SessionContext::new(); + ctx.register_parquet("t", parquet_path.to_str().unwrap(), Default::default()) + .await + .unwrap(); + let plan = ctx + .sql("SELECT * FROM t WHERE a = 2") + .await + .unwrap() + .create_physical_plan() + .await + .unwrap(); + let cache = make_cache(tmp_dir.path()).await; + let rewritten = rewrite_data_source_plan(plan, &cache); + let metrics = liquid_source(&rewritten).metrics().clone(); + + let batches = collect(rewritten, ctx.task_ctx()).await.unwrap(); + assert_eq!(batches.iter().map(RecordBatch::num_rows).sum::(), 1); + + let metric = metrics + .clone_inner() + .sum_by_name("row_groups_pruned_bloom_filter") + .unwrap(); + let datafusion::physical_plan::metrics::MetricValue::PruningMetrics { + pruning_metrics, .. + } = metric + else { + panic!("unexpected metric: {metric:?}"); + }; + assert_eq!(pruning_metrics.pruned(), 1); } async fn build_cache() -> LiquidCacheParquetRef { diff --git a/src/datafusion/src/optimizers/squeeze_hint.rs b/src/datafusion/src/optimizers/squeeze_hint.rs index 77c750074..b64cce3a3 100644 --- a/src/datafusion/src/optimizers/squeeze_hint.rs +++ b/src/datafusion/src/optimizers/squeeze_hint.rs @@ -223,11 +223,19 @@ impl HintAnalyzer { let usages = lineage_for_expr(&expr, &child); self.record(&usages); } - for aggr in agg.aggr_expr() { + for (aggr, filter) in agg.aggr_expr().iter().zip(agg.filter_expr()) { for expr in aggr.expressions() { let usages = lineage_for_expr(&expr, &child); self.record(&usages); } + for order_by in aggr.order_bys() { + let usages = lineage_for_expr(&order_by.expr, &child); + self.record(&usages); + } + if let Some(filter) = filter { + let usages = lineage_for_expr(filter, &child); + self.record(&usages); + } } return opaque(plan); } @@ -778,6 +786,30 @@ mod tests { assert_eq!(hints.get("date"), None); } + #[tokio::test] + async fn aggregate_filter_records_raw_column_use() { + let hints = hints_for( + "SELECT AVG(EXTRACT(YEAR FROM date)) \ + FILTER (WHERE date = DATE '2021-01-01') FROM t", + ) + .await; + + // The aggregate argument needs only YEAR, but its filter needs the + // exact date, so squeezing the column to YEAR would change the result. + assert_eq!(hints.get("date"), None); + } + + #[tokio::test] + async fn aggregate_order_by_records_raw_column_use() { + let hints = + hints_for("SELECT FIRST_VALUE(EXTRACT(MONTH FROM date) ORDER BY date DESC) FROM t") + .await; + + // The aggregate value needs only MONTH, but chronological ordering + // needs the exact date. + assert_eq!(hints.get("date"), None); + } + #[tokio::test] async fn substring_search_in_filter() { let hints = hints_for("SELECT date FROM t WHERE url LIKE '%example%'").await; diff --git a/src/datafusion/src/reader/plantime/mod.rs b/src/datafusion/src/reader/plantime/mod.rs index 068591249..0de78f046 100644 --- a/src/datafusion/src/reader/plantime/mod.rs +++ b/src/datafusion/src/reader/plantime/mod.rs @@ -3,10 +3,10 @@ pub(crate) use source::CachedMetaReaderFactory; pub use source::LiquidParquetSource; pub(crate) use source::ParquetMetadataCacheReader; -mod opener; +mod morselizer; mod row_filter; -mod row_group_filter; mod source; +pub(crate) use morselizer::{LiquidFileMetrics, LiquidMorselizer}; pub(crate) use row_filter::unevaluable_conjunct; pub use row_filter::{FilterCandidateBuilder, LiquidPredicate, LiquidRowFilter}; diff --git a/src/datafusion/src/reader/plantime/morselizer.rs b/src/datafusion/src/reader/plantime/morselizer.rs new file mode 100644 index 000000000..621728b41 --- /dev/null +++ b/src/datafusion/src/reader/plantime/morselizer.rs @@ -0,0 +1,1693 @@ +use std::{collections::VecDeque, fmt, future::Future, sync::Arc}; + +use arrow_schema::SchemaRef; +use datafusion::{ + common::{exec_err, internal_err}, + datasource::{ + listing::{FileRange, PartitionedFile}, + physical_plan::{ + ParquetFileMetrics, + parquet::{ + BloomFilterStatistics, PagePruningAccessPlanFilter, ParquetAccessPlan, + RowGroupAccessPlanFilter, + }, + }, + table_schema::TableSchema, + }, + error::Result, + physical_expr::{ + PhysicalExpr, PhysicalExprSimplifier, projection::ProjectionExprs, + utils::reassign_expr_columns, + }, + physical_expr_adapter::{PhysicalExprAdapterFactory, replace_columns_with_literals}, + physical_optimizer::pruning::{FilePruner, PruningPredicate, build_pruning_predicate}, + physical_plan::metrics::{Count, ExecutionPlanMetricsSet, MetricBuilder}, +}; +#[cfg(test)] +use datafusion_datasource::morsel::Morsel; +use datafusion_datasource::morsel::{MorselPlan, MorselPlanner, Morselizer}; +use futures::{FutureExt, future::BoxFuture}; +use log::debug; +use parquet::{ + arrow::{ + ParquetRecordBatchStreamBuilder, ProjectionMask, + arrow_reader::{ArrowReaderMetadata, ArrowReaderOptions, RowSelection}, + parquet_column, + }, + file::metadata::PageIndexPolicy, +}; + +use super::source::{CachedMetaReaderFactory, ParquetMetadataCacheReader}; +use crate::{ + cache::{ + BatchID, ColumnSqueezeHints, InsertArrowArrayError, LiquidCacheParquetRef, + ParquetFileIdentity, PrefetchOutcome, RowGroupSnapshots, + }, + reader::{ + plantime::row_filter::build_row_filter, + runtime::{ + LiquidRowGroupPlanner, apply_predicates, build_projection_schema, get_root_column_ids, + take_next_batch, + }, + }, + utils::row_selector_to_boolean_buffer, +}; +#[cfg(test)] +use liquid_cache::cache::{CachedBatchType, EntryID, LiquidCache}; + +pub(crate) struct LiquidMorselizer { + pub(crate) partition_index: usize, + pub(crate) projection: ProjectionExprs, + pub(crate) batch_size: usize, + pub(crate) predicate: Option>, + pub(crate) table_schema: TableSchema, + pub(crate) metrics: ExecutionPlanMetricsSet, + pub(crate) parquet_file_reader_factory: Arc, + pub(crate) reorder_filters: bool, + pub(crate) liquid_cache: LiquidCacheParquetRef, + pub(crate) expr_adapter_factory: Arc, + pub(crate) span: Option>, + pub(crate) squeeze_hints: Arc, + pub(crate) prefetch: bool, +} + +impl fmt::Debug for LiquidMorselizer { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.debug_struct("LiquidMorselizer") + .field("partition_index", &self.partition_index) + .field("batch_size", &self.batch_size) + .finish_non_exhaustive() + } +} + +impl Morselizer for LiquidMorselizer { + fn plan_file(&self, partitioned_file: PartitionedFile) -> Result> { + let file_range = partitioned_file.range.clone(); + let access_plan = partitioned_file.extensions.get_arc::(); + let file_name = partitioned_file.object_meta.location.to_string(); + let metrics = LiquidFileMetrics::new(self.partition_index, &file_name, &self.metrics); + let metadata_size_hint = partitioned_file.metadata_size_hint; + let file_identity = ParquetFileIdentity::new( + self.parquet_file_reader_factory.object_store_url().clone(), + partitioned_file.object_meta.location.to_string(), + ); + let reader = self.parquet_file_reader_factory.create_liquid_reader( + self.partition_index, + partitioned_file.clone(), + metadata_size_hint, + &self.metrics, + ); + + let logical_file_schema = Arc::clone(self.table_schema.file_schema()); + let output_schema = Arc::new( + self.projection + .project_schema(self.table_schema.table_schema())?, + ); + let mut projection = self.projection.clone(); + let mut predicate = self.predicate.clone(); + let mut literal_columns = std::collections::HashMap::new(); + for (field, value) in self + .table_schema + .table_partition_cols() + .iter() + .zip(&partitioned_file.partition_values) + { + literal_columns.insert(field.name().clone(), value.clone()); + } + if !literal_columns.is_empty() { + projection = projection.try_map_exprs(|expr| { + replace_columns_with_literals(Arc::clone(&expr), &literal_columns) + })?; + predicate = predicate + .map(|predicate| replace_columns_with_literals(predicate, &literal_columns)) + .transpose()?; + } + + // `FilePruner::try_new` itself decides whether a pruner is worth + // building: it returns `None` for a purely static predicate over a file + // with no usable column statistics. + let file_pruner = predicate.as_ref().and_then(|predicate| { + FilePruner::try_new( + Arc::clone(predicate), + &logical_file_schema, + &partitioned_file, + metrics.predicate_creation_errors.clone(), + ) + }); + let span = self.span.as_ref().map(|span| { + Arc::new(fastrace::Span::enter_with_parent( + format!("file_{file_name}"), + span, + )) + }); + + Ok(Box::new(LiquidFilePlanner { + state: LiquidOpenState::PruneFile(Box::new(PreparedLiquidOpen { + file_range, + access_plan, + file_name, + metrics, + file_pruner, + reader, + batch_size: self.batch_size, + logical_file_schema, + output_schema, + projection, + predicate, + reorder_filters: self.reorder_filters, + liquid_cache: self.liquid_cache.clone(), + expr_adapter_factory: Arc::clone(&self.expr_adapter_factory), + file_identity, + span, + squeeze_hints: Arc::clone(&self.squeeze_hints), + prefetch: self.prefetch, + })), + })) + } +} + +#[derive(Clone)] +pub(crate) struct LiquidFileMetrics { + pub(crate) file_metrics: ParquetFileMetrics, + pub(crate) predicate_creation_errors: Count, + pub(crate) batches_prefetched: Count, + pub(crate) prefetch_skipped: Count, +} + +impl LiquidFileMetrics { + pub(crate) fn new( + partition_index: usize, + file_name: &str, + metrics: &ExecutionPlanMetricsSet, + ) -> Self { + Self { + file_metrics: ParquetFileMetrics::new(partition_index, file_name, metrics), + predicate_creation_errors: MetricBuilder::new(metrics) + .global_counter("num_predicate_creation_errors"), + batches_prefetched: MetricBuilder::new(metrics) + .counter("batches_prefetched", partition_index), + prefetch_skipped: MetricBuilder::new(metrics) + .counter("prefetch_skipped", partition_index), + } + } +} + +struct PreparedLiquidOpen { + file_range: Option, + access_plan: Option>, + file_name: String, + metrics: LiquidFileMetrics, + file_pruner: Option, + reader: ParquetMetadataCacheReader, + batch_size: usize, + logical_file_schema: SchemaRef, + output_schema: SchemaRef, + projection: ProjectionExprs, + predicate: Option>, + reorder_filters: bool, + liquid_cache: LiquidCacheParquetRef, + expr_adapter_factory: Arc, + file_identity: ParquetFileIdentity, + span: Option>, + squeeze_hints: Arc, + prefetch: bool, +} + +struct MetadataLoadedLiquidOpen { + prepared: Box, + reader_metadata: ArrowReaderMetadata, + options: ArrowReaderOptions, +} + +struct PreparedRowGroups { + context: RowGroupPlanningContext, + row_groups: RowGroupAccessPlanFilter, +} + +struct RowGroupPlanningContext { + prepared: Box, + reader_metadata: ArrowReaderMetadata, + physical_file_schema: SchemaRef, + cache_full_schema: SchemaRef, + builder: ParquetRecordBatchStreamBuilder, + projection_mask: ProjectionMask, + row_filter: Option, + pruning_predicate: Option>, + page_pruning_predicate: Option>, +} + +struct BloomFiltersLoadedLiquidOpen { + prepared: PreparedRowGroups, + bloom_filters: Vec, +} + +struct PlannedRowGroups { + context: RowGroupPlanningContext, + access_plan: ParquetAccessPlan, +} + +enum LiquidOpenState { + PruneFile(Box), + LoadMetadata(BoxFuture<'static, Result>), + PrepareAndPruneByStats(Box), + LoadBloomFilters(BoxFuture<'static, Result>), + PruneBloomAndPages(Box), + PlanRowGroups(Box), + Done, +} + +impl fmt::Debug for LiquidOpenState { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.write_str(match self { + Self::PruneFile(_) => "PruneFile", + Self::LoadMetadata(_) => "LoadMetadata", + Self::PrepareAndPruneByStats(_) => "PrepareAndPruneByStats", + Self::LoadBloomFilters(_) => "LoadBloomFilters", + Self::PruneBloomAndPages(_) => "PruneBloomAndPages", + Self::PlanRowGroups(_) => "PlanRowGroups", + Self::Done => "Done", + }) + } +} + +impl LiquidOpenState { + fn transition(self) -> Result { + match self { + Self::PruneFile(mut prepared) => { + if let Some(file_pruner) = &mut prepared.file_pruner + && file_pruner.should_prune()? + { + prepared + .metrics + .file_metrics + .files_ranges_pruned_statistics + .add_pruned(1); + return Ok(Self::Done); + } + + prepared + .metrics + .file_metrics + .files_ranges_pruned_statistics + .add_matched(1); + Ok(Self::LoadMetadata( + async move { + let options = ArrowReaderOptions::new() + // `Optional`, not `Required`: the page index is an + // optimization, not a correctness requirement. It drives + // page-level pruning, which no-ops when the index is + // absent. `Required` instead fails the whole read with + // `missing offset index` on any file that advertises a + // page index while one of its column chunks carries no + // offset index -- a shape valid parquet is free to have. + .with_page_index_policy(PageIndexPolicy::Optional); + let metadata_load_time = + prepared.metrics.file_metrics.metadata_load_time.clone(); + let mut timer = metadata_load_time.timer(); + let reader_metadata = + ArrowReaderMetadata::load_async(&mut prepared.reader, options.clone()) + .await?; + timer.stop(); + Ok(MetadataLoadedLiquidOpen { + prepared, + reader_metadata, + options, + }) + } + .boxed(), + )) + } + Self::LoadMetadata(future) => Ok(Self::LoadMetadata(future)), + Self::PrepareAndPruneByStats(loaded) => prepare_and_prune_by_stats(*loaded), + Self::LoadBloomFilters(future) => Ok(Self::LoadBloomFilters(future)), + Self::PruneBloomAndPages(loaded) => { + let mut prepared = loaded.prepared; + let predicate = prepared + .context + .pruning_predicate + .as_deref() + .expect("bloom filters are loaded only with a pruning predicate"); + prepared.row_groups.prune_by_bloom_filters( + predicate, + &prepared.context.prepared.metrics.file_metrics, + &loaded.bloom_filters, + ); + Ok(Self::PlanRowGroups(Box::new(prune_pages(prepared)))) + } + Self::PlanRowGroups(planned) => Ok(Self::PlanRowGroups(planned)), + Self::Done => Ok(Self::Done), + } + } +} + +fn prepare_and_prune_by_stats(mut loaded: MetadataLoadedLiquidOpen) -> Result { + let metadata_load_time = loaded + .prepared + .metrics + .file_metrics + .metadata_load_time + .clone(); + let mut metadata_timer = metadata_load_time.timer(); + let physical_file_schema = Arc::clone(loaded.reader_metadata.schema()); + let cache_full_schema = Arc::clone(&physical_file_schema); + loaded.options = loaded + .options + .with_schema(Arc::clone(&physical_file_schema)); + loaded.reader_metadata = ArrowReaderMetadata::try_new( + Arc::clone(loaded.reader_metadata.metadata()), + loaded.options, + )?; + debug_assert!( + Arc::strong_count(loaded.reader_metadata.metadata()) > 1, + "meta data must be cached already" + ); + + let rewriter = loaded.prepared.expr_adapter_factory.create( + Arc::clone(&loaded.prepared.logical_file_schema), + Arc::clone(&physical_file_schema), + )?; + let simplifier = PhysicalExprSimplifier::new(&physical_file_schema); + loaded.prepared.predicate = loaded + .prepared + .predicate + .take() + .map(|predicate| simplifier.simplify(rewriter.rewrite(predicate)?)) + .transpose()?; + loaded.prepared.projection = loaded + .prepared + .projection + .try_map_exprs(|expr| simplifier.simplify(rewriter.rewrite(expr)?))?; + + let (pruning_predicate, page_pruning_predicate) = build_pruning_predicates( + loaded.prepared.predicate.as_ref(), + &physical_file_schema, + &loaded.prepared.metrics.predicate_creation_errors, + ); + metadata_timer.stop(); + let builder = ParquetRecordBatchStreamBuilder::new_with_metadata( + loaded.prepared.reader.clone(), + loaded.reader_metadata.clone(), + ); + let projection_mask = ProjectionMask::roots( + builder.parquet_schema(), + loaded.prepared.projection.column_indices(), + ); + // A failure here is not recoverable by ignoring it. DataFusion removed the + // `FilterExec` when it pushed this predicate down, so the row filter is the + // only place the predicate is applied; carrying on without one returns rows + // the query excluded. Fail the query instead (issue #23). + let row_filter = match loaded.prepared.predicate.as_ref() { + Some(predicate) => build_row_filter( + predicate, + &physical_file_schema, + loaded.reader_metadata.metadata(), + loaded.prepared.reorder_filters, + &loaded.prepared.metrics.file_metrics, + )?, + None => None, + }; + + let metadata = builder.metadata(); + let row_group_metadata = metadata.row_groups(); + let access_plan = create_initial_plan( + &loaded.prepared.file_name, + loaded.prepared.access_plan.take(), + row_group_metadata.len(), + )?; + let mut row_groups = RowGroupAccessPlanFilter::new(access_plan); + if let Some(range) = &loaded.prepared.file_range { + row_groups.prune_by_range(row_group_metadata, range); + } + if let Some(predicate) = pruning_predicate.as_deref() { + row_groups.prune_by_statistics( + &physical_file_schema, + builder.parquet_schema(), + row_group_metadata, + predicate, + &loaded.prepared.metrics.file_metrics, + ); + } + + let prepared = PreparedRowGroups { + context: RowGroupPlanningContext { + prepared: loaded.prepared, + reader_metadata: loaded.reader_metadata, + physical_file_schema, + cache_full_schema, + builder, + projection_mask, + row_filter, + pruning_predicate, + page_pruning_predicate, + }, + row_groups, + }; + if prepared.context.pruning_predicate.is_some() && !prepared.row_groups.is_empty() { + Ok(LiquidOpenState::LoadBloomFilters( + async move { + let mut prepared = prepared; + let predicate = Arc::clone( + prepared + .context + .pruning_predicate + .as_ref() + .expect("pruning predicate was checked before scheduling bloom I/O"), + ); + let bloom_filters = load_bloom_filters( + &mut prepared.context.builder, + predicate.as_ref(), + &prepared.context.prepared.metrics.file_metrics, + &prepared.row_groups, + ) + .await; + Ok(BloomFiltersLoadedLiquidOpen { + prepared, + bloom_filters, + }) + } + .boxed(), + )) + } else { + Ok(LiquidOpenState::PlanRowGroups(Box::new(prune_pages( + prepared, + )))) + } +} + +fn prune_pages(prepared: PreparedRowGroups) -> PlannedRowGroups { + let PreparedRowGroups { + context, + row_groups, + } = prepared; + let mut access_plan = row_groups.build(); + if !access_plan.is_empty() + && let Some(predicate) = &context.page_pruning_predicate + { + access_plan = predicate.prune_plan_with_page_index( + access_plan, + &context.physical_file_schema, + context.builder.parquet_schema(), + context.builder.metadata().as_ref(), + &context.prepared.metrics.file_metrics, + ); + } + PlannedRowGroups { + context, + access_plan, + } +} + +struct LiquidFilePlanner { + state: LiquidOpenState, +} + +impl fmt::Debug for LiquidFilePlanner { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.debug_tuple("LiquidFilePlanner") + .field(&self.state) + .finish() + } +} + +impl LiquidFilePlanner { + fn schedule_io(future: F) -> MorselPlan + where + F: Future> + Send + 'static, + { + let future = async move { + let state = future.await?; + Ok(Box::new(Self { state }) as Box) + }; + MorselPlan::new().with_pending_planner(future) + } +} + +impl MorselPlanner for LiquidFilePlanner { + fn plan(self: Box) -> Result> { + let state = self.state.transition()?; + match state { + LiquidOpenState::LoadMetadata(future) => Ok(Some(Self::schedule_io(async move { + Ok(LiquidOpenState::PrepareAndPruneByStats(Box::new( + future.await?, + ))) + }))), + LiquidOpenState::LoadBloomFilters(future) => Ok(Some(Self::schedule_io(async move { + Ok(LiquidOpenState::PruneBloomAndPages(Box::new(future.await?))) + }))), + LiquidOpenState::PlanRowGroups(planned) => plan_row_group_morsels(*planned), + LiquidOpenState::Done => Ok(None), + cpu_state => Ok(Some( + MorselPlan::new().with_planners(vec![Box::new(Self { state: cpu_state })]), + )), + } + } +} + +fn plan_row_group_morsels(planned: PlannedRowGroups) -> Result> { + let PlannedRowGroups { + context, + access_plan, + } = planned; + let prefetch = context.prepared.prefetch; + let cached_file = context + .prepared + .liquid_cache + .register_or_get_file_with_hints( + context.prepared.file_identity.clone(), + Arc::clone(&context.cache_full_schema), + Arc::clone(&context.prepared.squeeze_hints), + ); + let metadata = Arc::clone(context.reader_metadata.metadata()); + let schema_descriptor = metadata.file_metadata().schema_descr(); + let projection_column_ids = get_root_column_ids(schema_descriptor, &context.projection_mask); + let stream_schema = build_projection_schema(&cached_file.schema(), &projection_column_ids); + let replace_schema = !stream_schema.eq(&context.prepared.output_schema); + let projection = context + .prepared + .projection + .try_map_exprs(|expr| reassign_expr_columns(expr, &stream_schema))?; + let projector = Arc::new(projection.make_projector(&stream_schema)?); + let row_group_planner = Arc::new(LiquidRowGroupPlanner { + metadata: Arc::clone(&metadata), + input: context.prepared.reader.clone(), + row_filter: context.row_filter, + cached_file, + projection: context.projection_mask, + batch_size: context.prepared.batch_size, + stream_schema, + output_schema: Arc::clone(&context.prepared.output_schema), + projector, + replace_schema, + span: context.prepared.span, + liquid_cache: context.prepared.liquid_cache, + metrics: context.prepared.metrics.clone(), + }); + + let row_group_indexes = access_plan.row_group_indexes(); + let row_group_metadata = metadata.row_groups(); + let mut selection = access_plan.into_overall_row_selection(row_group_metadata)?; + let mut queue = VecDeque::with_capacity(row_group_indexes.len()); + for row_group_idx in row_group_indexes { + let row_count = row_group_metadata[row_group_idx].num_rows() as usize; + let row_group_selection = selection + .as_mut() + .map(|selection| selection.split_off(row_count)); + let row_group_selection = row_group_selection.unwrap_or_else(|| { + vec![parquet::arrow::arrow_reader::RowSelector::select(row_count)].into() + }); + if row_group_selection.row_count() > 0 { + queue.push_back((row_group_idx, row_group_selection)); + } + } + + if queue.is_empty() { + return Ok(None); + } + + let chain = LiquidRowGroupChain { + planner: row_group_planner, + queue, + snapshots: Arc::default(), + prefetch, + }; + let plan = if prefetch { + MorselPlan::new().with_pending_planner(prefetch_future(chain)) + } else { + MorselPlan::new().with_planners(vec![Box::new(chain)]) + }; + Ok(Some(plan)) +} + +struct LiquidRowGroupChain { + planner: Arc, + queue: VecDeque<(usize, RowSelection)>, + snapshots: Arc, + prefetch: bool, +} + +impl fmt::Debug for LiquidRowGroupChain { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.debug_struct("LiquidRowGroupChain") + .field("remaining_row_groups", &self.queue.len()) + .finish_non_exhaustive() + } +} + +impl MorselPlanner for LiquidRowGroupChain { + fn plan(mut self: Box) -> Result> { + let (row_group_idx, selection) = self + .queue + .pop_front() + .expect("a row group chain is never empty"); + let snapshots = std::mem::take(&mut self.snapshots); + let Some(morsel) = self.planner.plan(row_group_idx, Some(selection), snapshots) else { + return internal_err!("selected row group {row_group_idx} produced no morsel"); + }; + let mut plan = MorselPlan::new().with_morsels(vec![Box::new(morsel)]); + if self.queue.is_empty() { + return Ok(Some(plan)); + } + + if !self.prefetch { + return Ok(Some(plan.with_planners(vec![self]))); + } + + let next_row_group = self.queue.front().unwrap().0; + let estimate = self.planner.estimated_bytes(next_row_group); + let headroom = self + .planner + .liquid_cache + .max_memory_bytes() + .saturating_sub(self.planner.liquid_cache.memory_usage_bytes()); + if headroom >= estimate { + plan = plan.with_pending_planner(prefetch_future(*self)); + } else { + self.planner.metrics.prefetch_skipped.add(1); + plan = plan.with_planners(vec![self]); + } + Ok(Some(plan)) + } +} + +async fn prefetch_future(chain: LiquidRowGroupChain) -> Result> { + Ok(Box::new(prefetch_front(chain).await) as Box) +} + +async fn prefetch_front(chain: LiquidRowGroupChain) -> LiquidRowGroupChain { + let (row_group_idx, selection) = chain.queue.front().expect("prefetch chain has work"); + let mut selectors: VecDeque<_> = selection.clone().into(); + let batch_size = chain.planner.cached_file.batch_size(); + let selected_batches = std::iter::from_fn(|| take_next_batch(&mut selectors, batch_size)) + .enumerate() + .filter_map(|(idx, selection)| { + let selection = row_selector_to_boolean_buffer(&selection); + (selection.count_set_bits() > 0).then_some((BatchID::from_raw(idx as u16), selection)) + }) + .collect::>(); + let estimate = chain.planner.estimated_bytes(*row_group_idx); + let per_batch_estimate = estimate / selected_batches.len().max(1); + let snapshots = Arc::clone(&chain.snapshots); + let mut context = chain + .planner + .prefetch_context(*row_group_idx, Arc::clone(&snapshots)); + + for (batch_id, input_selection) in selected_batches { + let mut produced_snapshots = false; + let predicate_summary = prefetch_columns( + &context.cached_row_group, + batch_id, + &context.predicate_column_ids, + ) + .await; + produced_snapshots |= predicate_summary.any_snapshotted; + + if predicate_summary.any_missing { + match materialize_prefetch_batch(&mut context, batch_id).await { + Ok(()) => produced_snapshots = true, + Err(error) => { + debug!("Stopping row group {row_group_idx} prefetch: {error}"); + break; + } + } + } + + let filtered_selection = if let Some(filter) = context.row_filter.as_mut() { + match apply_predicates(&context.cached_row_group, batch_id, input_selection, filter) + .await + { + Ok(selection) => selection, + Err(error) => { + debug!("Stopping row group {row_group_idx} prefetch: {error}"); + break; + } + } + } else { + Some(input_selection) + }; + + if let Some(filtered_selection) = filtered_selection { + snapshots.insert_selection(batch_id, filtered_selection.clone()); + + if filtered_selection.count_set_bits() > 0 { + let projection_summary = prefetch_columns( + &context.cached_row_group, + batch_id, + &context.projection_column_ids, + ) + .await; + produced_snapshots |= projection_summary.any_snapshotted; + if projection_summary.any_missing { + match materialize_prefetch_batch(&mut context, batch_id).await { + Ok(()) => produced_snapshots = true, + Err(error) => { + debug!("Stopping row group {row_group_idx} prefetch: {error}"); + break; + } + } + } + } + } + + if produced_snapshots { + chain.planner.metrics.batches_prefetched.add(1); + } + let headroom = chain + .planner + .liquid_cache + .max_memory_bytes() + .saturating_sub(chain.planner.liquid_cache.memory_usage_bytes()); + if headroom < per_batch_estimate { + break; + } + } + + chain +} + +struct PrefetchColumnsSummary { + any_missing: bool, + any_snapshotted: bool, +} + +async fn prefetch_columns( + row_group: &crate::cache::CachedRowGroupRef, + batch_id: BatchID, + column_ids: &[usize], +) -> PrefetchColumnsSummary { + let mut summary = PrefetchColumnsSummary { + any_missing: false, + any_snapshotted: false, + }; + for column_id in column_ids { + let column = row_group.get_column(*column_id as u64).unwrap(); + match column.prefetch_snapshot(batch_id).await { + PrefetchOutcome::Snapshotted => summary.any_snapshotted = true, + PrefetchOutcome::Missing => summary.any_missing = true, + PrefetchOutcome::AlreadySnapshotted | PrefetchOutcome::Squeezed => {} + } + } + summary +} + +async fn materialize_prefetch_batch( + context: &mut crate::reader::runtime::LiquidRowGroupPrefetchContext, + batch_id: BatchID, +) -> std::result::Result<(), parquet::errors::ParquetError> { + let record_batch = context.fallback.fetch_batch(batch_id).await?; + for (position, column_id) in context.cache_column_ids.iter().enumerate() { + let column = context + .cached_row_group + .get_column(*column_id as u64) + .unwrap(); + let array = Arc::clone(record_batch.column(position)); + match column.insert(batch_id, Arc::clone(&array)).await { + Ok(()) | Err(InsertArrowArrayError::AlreadyCached) => {} + Err(InsertArrowArrayError::CacheFull) => {} + } + column.insert_snapshot(batch_id, array); + } + Ok(()) +} + +async fn load_bloom_filters( + builder: &mut ParquetRecordBatchStreamBuilder, + predicate: &PruningPredicate, + file_metrics: &ParquetFileMetrics, + row_groups: &RowGroupAccessPlanFilter, +) -> Vec { + let mut row_group_bloom_filters = + vec![BloomFilterStatistics::new(); builder.metadata().num_row_groups()]; + let parquet_columns = predicate + .literal_columns() + .into_iter() + .filter_map(|column_name| { + let parquet_schema = builder.parquet_schema(); + let (column_idx, _) = parquet_column(parquet_schema, predicate.schema(), &column_name)?; + let column = parquet_schema.column(column_idx); + Some(( + column_name, + column_idx, + column.physical_type(), + column.type_length(), + )) + }) + .collect::>(); + + for row_group_idx in row_groups.row_group_indexes() { + let mut bloom_filters = BloomFilterStatistics::with_capacity(parquet_columns.len()); + for (column_name, column_idx, physical_type, type_length) in &parquet_columns { + let bloom_filter = match builder + .get_row_group_column_bloom_filter(row_group_idx, *column_idx) + .await + { + Ok(Some(bloom_filter)) => bloom_filter, + Ok(None) => continue, + Err(error) => { + debug!("Ignoring error reading bloom filter: {error}"); + file_metrics.predicate_evaluation_errors.add(1); + continue; + } + }; + bloom_filters.insert(column_name, bloom_filter, *physical_type, *type_length); + } + row_group_bloom_filters[row_group_idx] = bloom_filters; + } + + row_group_bloom_filters +} + +fn create_initial_plan( + file_name: &str, + access_plan: Option>, + row_group_count: usize, +) -> Result { + if let Some(access_plan) = access_plan { + let plan_len = access_plan.len(); + if plan_len != row_group_count { + return exec_err!( + "Invalid ParquetAccessPlan for {file_name}. Specified {plan_len} row groups, but file has {row_group_count}" + ); + } + return Ok(access_plan.as_ref().clone()); + } + + Ok(ParquetAccessPlan::new_all(row_group_count)) +} + +pub(crate) fn build_pruning_predicates( + predicate: Option<&Arc>, + file_schema: &SchemaRef, + predicate_creation_errors: &Count, +) -> ( + Option>, + Option>, +) { + let Some(predicate) = predicate else { + return (None, None); + }; + let pruning_predicate = build_pruning_predicate( + Arc::clone(predicate), + file_schema, + predicate_creation_errors, + ); + let page_pruning_predicate = build_page_pruning_predicate(predicate, file_schema); + (pruning_predicate, Some(page_pruning_predicate)) +} + +pub(crate) fn build_page_pruning_predicate( + predicate: &Arc, + file_schema: &SchemaRef, +) -> Arc { + Arc::new(PagePruningAccessPlanFilter::new( + predicate, + Arc::clone(file_schema), + )) +} + +#[cfg(test)] +mod tests { + use std::{ + collections::VecDeque, + fs::File, + sync::atomic::{AtomicUsize, Ordering}, + }; + + use arrow::{ + array::{Array, ArrayRef, Int32Array, RecordBatch}, + datatypes::{DataType, Field, Schema}, + }; + use datafusion::{ + common::ScalarValue, + datasource::{ + listing::PartitionedFile, + physical_plan::{FileScanConfigBuilder, FileSource, ParquetSource}, + }, + execution::object_store::ObjectStoreUrl, + logical_expr::Operator, + physical_expr::{ + PhysicalExpr, + expressions::{BinaryExpr, Column, Literal}, + projection::ProjectionExprs, + }, + physical_expr_adapter::DefaultPhysicalExprAdapterFactory, + physical_plan::metrics::ExecutionPlanMetricsSet, + }; + use futures::StreamExt; + use liquid_cache::{ + cache::{AlwaysHydrate, squeeze_policies::Evict}, + cache_policies::LiquidPolicy, + }; + use object_store::local::LocalFileSystem; + use parquet::arrow::{ArrowWriter, async_reader::AsyncFileReader}; + + use crate::{ + cache::{BatchID, CachedFileRef, CachedRowGroupRef, LiquidCacheParquet}, + reader::{LiquidParquetSource, extract_multi_column_or}, + }; + + use super::*; + + static NEXT_FILE_ID: AtomicUsize = AtomicUsize::new(0); + + struct PlannedTestFile { + morsels: Vec>, + _cache: Arc, + cached_file: CachedFileRef, + _tmp_dir: tempfile::TempDir, + } + + struct TestFilePlanner { + planner: Box, + cache: Arc, + cached_file: CachedFileRef, + tmp_dir: tempfile::TempDir, + } + + fn schema() -> SchemaRef { + Arc::new(Schema::new(vec![ + Field::new("a", DataType::Int32, false), + Field::new("b", DataType::Int32, false), + ])) + } + + fn write_two_row_group_file(path: &std::path::Path, schema: SchemaRef) { + let file = File::create(path).unwrap(); + let mut writer = ArrowWriter::try_new(file, Arc::clone(&schema), None).unwrap(); + writer + .write( + &RecordBatch::try_new( + Arc::clone(&schema), + vec![ + Arc::new(Int32Array::from(vec![0, 1, 2, 3])), + Arc::new(Int32Array::from(vec![10, 11, 12, 13])), + ], + ) + .unwrap(), + ) + .unwrap(); + writer.flush().unwrap(); + writer + .write( + &RecordBatch::try_new( + schema, + vec![ + Arc::new(Int32Array::from(vec![4, 5, 6, 7])), + Arc::new(Int32Array::from(vec![14, 15, 16, 17])), + ], + ) + .unwrap(), + ) + .unwrap(); + writer.close().unwrap(); + } + + fn write_single_row_group_file(path: &std::path::Path, schema: SchemaRef, a: Vec) { + let file = File::create(path).unwrap(); + let b = a.iter().map(|value| value + 1000).collect::>(); + let batch = RecordBatch::try_new( + Arc::clone(&schema), + vec![Arc::new(Int32Array::from(a)), Arc::new(Int32Array::from(b))], + ) + .unwrap(); + let mut writer = ArrowWriter::try_new(file, schema, None).unwrap(); + writer.write(&batch).unwrap(); + writer.close().unwrap(); + } + + async fn drive_planner(planner: Box) -> Vec> { + let mut planners = VecDeque::from([planner]); + let mut morsels = Vec::new(); + while let Some(planner) = planners.pop_front() { + let Some(mut plan) = planner.plan().unwrap() else { + continue; + }; + morsels.extend(plan.take_morsels()); + planners.extend(plan.take_ready_planners()); + if let Some(pending) = plan.take_pending_planner() { + planners.push_back(pending.await.unwrap()); + } + } + morsels + } + + struct PlanOptions { + max_memory_bytes: usize, + max_disk_bytes: usize, + predicate: Option>, + projection_columns: Vec, + single_row_group_values: Option>, + } + + impl Default for PlanOptions { + fn default() -> Self { + Self { + max_memory_bytes: usize::MAX, + max_disk_bytes: usize::MAX, + predicate: None, + projection_columns: vec![0, 1], + single_row_group_values: None, + } + } + } + + async fn create_test_cache( + path: &std::path::Path, + max_memory_bytes: usize, + max_disk_bytes: usize, + ) -> Arc { + let store = crate::test_utils::mount_test_store(path).await; + Arc::new( + LiquidCacheParquet::new( + 4, + max_memory_bytes, + max_disk_bytes, + store, + Box::new(LiquidPolicy::new()), + Box::new(Evict), + Box::new(AlwaysHydrate::new()), + ) + .await, + ) + } + + async fn prepare_test_file(options: PlanOptions) -> TestFilePlanner { + let schema = schema(); + let tmp_dir = tempfile::tempdir().unwrap(); + let file_id = NEXT_FILE_ID.fetch_add(1, Ordering::Relaxed); + let file_name = "data.parquet".to_string(); + let parquet_path = tmp_dir.path().join(&file_name); + if let Some(values) = options.single_row_group_values { + write_single_row_group_file(&parquet_path, Arc::clone(&schema), values); + } else { + write_two_row_group_file(&parquet_path, Arc::clone(&schema)); + } + let partitioned_file = PartitionedFile::new( + file_name.clone(), + std::fs::metadata(&parquet_path).unwrap().len(), + ); + let object_store = Arc::new(LocalFileSystem::new_with_prefix(tmp_dir.path()).unwrap()); + let cache = create_test_cache( + tmp_dir.path(), + options.max_memory_bytes, + options.max_disk_bytes, + ) + .await; + let metrics = ExecutionPlanMetricsSet::new(); + let morselizer = LiquidMorselizer { + partition_index: 0, + projection: ProjectionExprs::from_indices(&options.projection_columns, schema.as_ref()), + batch_size: 4, + predicate: options.predicate, + table_schema: TableSchema::from(Arc::clone(&schema)), + metrics: metrics.clone(), + parquet_file_reader_factory: Arc::new(CachedMetaReaderFactory::new( + object_store, + ObjectStoreUrl::parse(format!("test-{file_id}:///")).unwrap(), + )), + reorder_filters: false, + liquid_cache: cache.clone(), + expr_adapter_factory: Arc::new(DefaultPhysicalExprAdapterFactory), + span: None, + squeeze_hints: Arc::default(), + prefetch: true, + }; + let cached_file = cache.register_or_get_file( + ParquetFileIdentity::new( + ObjectStoreUrl::parse(format!("test-{file_id}:///")).unwrap(), + file_name, + ), + schema, + ); + TestFilePlanner { + planner: morselizer.plan_file(partitioned_file).unwrap(), + cache, + cached_file, + tmp_dir, + } + } + + async fn plan_test_file(options: PlanOptions) -> PlannedTestFile { + let prepared = prepare_test_file(options).await; + let morsels = drive_planner(prepared.planner).await; + PlannedTestFile { + morsels, + _cache: prepared.cache, + cached_file: prepared.cached_file, + _tmp_dir: prepared.tmp_dir, + } + } + + async fn advance_to_row_group_chain( + mut planner: Box, + ) -> Box { + loop { + let mut plan = planner.plan().unwrap().expect("file has row groups"); + assert!(plan.take_morsels().is_empty()); + if let Some(ready) = plan.take_ready_planners().pop() { + planner = ready; + continue; + } + let pending = plan.take_pending_planner().expect("planner has more work"); + planner = pending.await.unwrap(); + if format!("{planner:?}").contains("LiquidRowGroupChain") { + return planner; + } + } + } + + fn gt_expr(column_name: &str, column_index: usize, literal: i32) -> Arc { + Arc::new(BinaryExpr::new( + Arc::new(Column::new(column_name, column_index)), + Operator::Gt, + Arc::new(Literal::new(ScalarValue::Int32(Some(literal)))), + )) + } + + fn eq_expr(column_name: &str, column_index: usize, literal: i32) -> Arc { + Arc::new(BinaryExpr::new( + Arc::new(Column::new(column_name, column_index)), + Operator::Eq, + Arc::new(Literal::new(ScalarValue::Int32(Some(literal)))), + )) + } + + #[tokio::test] + async fn metadata_cache_is_scoped_to_object_store() { + let schema = schema(); + let dir_a = tempfile::tempdir().unwrap(); + let dir_b = tempfile::tempdir().unwrap(); + let path_a = dir_a.path().join("data.parquet"); + let path_b = dir_b.path().join("data.parquet"); + write_single_row_group_file(&path_a, schema.clone(), vec![1]); + write_single_row_group_file(&path_b, schema, vec![1, 2]); + let metrics = ExecutionPlanMetricsSet::new(); + let mut reader_a = CachedMetaReaderFactory::new( + Arc::new(LocalFileSystem::new_with_prefix(dir_a.path()).unwrap()), + ObjectStoreUrl::parse("store-a:///").unwrap(), + ) + .create_liquid_reader( + 0, + PartitionedFile::new("data.parquet", std::fs::metadata(path_a).unwrap().len()), + None, + &metrics, + ); + let mut reader_b = CachedMetaReaderFactory::new( + Arc::new(LocalFileSystem::new_with_prefix(dir_b.path()).unwrap()), + ObjectStoreUrl::parse("store-b:///").unwrap(), + ) + .create_liquid_reader( + 0, + PartitionedFile::new("data.parquet", std::fs::metadata(path_b).unwrap().len()), + None, + &metrics, + ); + + let metadata_a = reader_a.get_metadata(None).await.unwrap(); + let metadata_b = reader_b.get_metadata(None).await.unwrap(); + + assert_eq!(metadata_a.file_metadata().num_rows(), 1); + assert_eq!(metadata_b.file_metadata().num_rows(), 2); + } + + #[tokio::test] + async fn data_cache_is_scoped_to_object_store() { + let schema = schema(); + let dir_a = tempfile::tempdir().unwrap(); + let dir_b = tempfile::tempdir().unwrap(); + let cache_dir = tempfile::tempdir().unwrap(); + let path_a = dir_a.path().join("data.parquet"); + let path_b = dir_b.path().join("data.parquet"); + write_single_row_group_file(&path_a, Arc::clone(&schema), vec![1, 2]); + write_single_row_group_file(&path_b, Arc::clone(&schema), vec![10, 20]); + + let cache = create_test_cache(cache_dir.path(), usize::MAX, usize::MAX).await; + let metrics = ExecutionPlanMetricsSet::new(); + let projection = ProjectionExprs::from_indices(&[0, 1], schema.as_ref()); + let morselizer_a = LiquidMorselizer { + partition_index: 0, + projection: projection.clone(), + batch_size: 4, + predicate: None, + table_schema: TableSchema::from(Arc::clone(&schema)), + metrics: metrics.clone(), + parquet_file_reader_factory: Arc::new(CachedMetaReaderFactory::new( + Arc::new(LocalFileSystem::new_with_prefix(dir_a.path()).unwrap()), + ObjectStoreUrl::parse("data-cache-a:///").unwrap(), + )), + reorder_filters: false, + liquid_cache: Arc::clone(&cache), + expr_adapter_factory: Arc::new(DefaultPhysicalExprAdapterFactory), + span: None, + squeeze_hints: Arc::default(), + prefetch: true, + }; + let morselizer_b = LiquidMorselizer { + partition_index: 0, + projection, + batch_size: 4, + predicate: None, + table_schema: TableSchema::from(Arc::clone(&schema)), + metrics, + parquet_file_reader_factory: Arc::new(CachedMetaReaderFactory::new( + Arc::new(LocalFileSystem::new_with_prefix(dir_b.path()).unwrap()), + ObjectStoreUrl::parse("data-cache-b:///").unwrap(), + )), + reorder_filters: false, + liquid_cache: cache, + expr_adapter_factory: Arc::new(DefaultPhysicalExprAdapterFactory), + span: None, + squeeze_hints: Arc::default(), + prefetch: true, + }; + + let file_a = PartitionedFile::new("data.parquet", std::fs::metadata(path_a).unwrap().len()); + let file_b = PartitionedFile::new("data.parquet", std::fs::metadata(path_b).unwrap().len()); + let rows_a = + collect_columns(drive_planner(morselizer_a.plan_file(file_a).unwrap()).await).await; + let rows_b = + collect_columns(drive_planner(morselizer_b.plan_file(file_b).unwrap()).await).await; + + assert_eq!(rows_a, (vec![1, 2], vec![1001, 1002])); + assert_eq!(rows_b, (vec![10, 20], vec![1010, 1020])); + } + + async fn collect_columns(morsels: Vec>) -> (Vec, Vec) { + let mut a = Vec::new(); + let mut b = Vec::new(); + for morsel in morsels { + let batches = morsel.into_stream().collect::>().await; + for batch in batches { + let batch = batch.unwrap(); + a.extend( + batch + .column(0) + .as_any() + .downcast_ref::() + .unwrap() + .values(), + ); + if batch.num_columns() > 1 { + b.extend( + batch + .column(1) + .as_any() + .downcast_ref::() + .unwrap() + .values(), + ); + } + } + } + (a, b) + } + + async fn insert_batches( + row_group: &CachedRowGroupRef, + column_id: usize, + batches: &[(u16, &[i32])], + ) { + let column = row_group.get_column(column_id as u64).unwrap(); + for (batch_idx, values) in batches { + let array: ArrayRef = Arc::new(Int32Array::from(values.to_vec())); + column + .insert(BatchID::from_raw(*batch_idx), array) + .await + .unwrap(); + } + } + + async fn contains(row_group: &CachedRowGroupRef, column_id: usize, batch_idx: u16) -> bool { + row_group + .get_column(column_id as u64) + .unwrap() + .get_arrow_array_test_only(BatchID::from_raw(batch_idx)) + .await + .is_some() + } + + fn kind_of(cache: &LiquidCache, id: &EntryID) -> Option { + let mut kind = None; + cache.for_each_entry(|entry_id, entry| { + if entry_id == id { + kind = Some(CachedBatchType::from(entry)); + } + }); + kind + } + + #[tokio::test] + async fn plans_one_morsel_per_selected_row_group() { + let all = plan_test_file(PlanOptions { + ..Default::default() + }) + .await; + assert_eq!(all.morsels.len(), 2); + assert_eq!( + collect_columns(all.morsels).await.0, + vec![0, 1, 2, 3, 4, 5, 6, 7] + ); + + let pruned = plan_test_file(PlanOptions { + predicate: Some(gt_expr("a", 0, 3)), + ..Default::default() + }) + .await; + assert_eq!(pruned.morsels.len(), 1); + assert_eq!(collect_columns(pruned.morsels).await.0, vec![4, 5, 6, 7]); + } + + #[tokio::test] + async fn prefetch_hands_snapshots_to_next_morsel() { + let file = prepare_test_file(PlanOptions::default()).await; + let chain = advance_to_row_group_chain(file.planner).await; + let mut first_plan = chain.plan().unwrap().unwrap(); + let mut morsels = first_plan.take_morsels(); + assert_eq!(morsels.len(), 1); + let next = first_plan.take_pending_planner().unwrap().await.unwrap(); + + let row_group = file.cached_file.create_row_group(1, vec![]); + for column_id in 0..2 { + let id = row_group + .get_column(column_id) + .unwrap() + .entry_id(BatchID::from_raw(0)) + .into(); + assert_eq!( + kind_of(file.cache.storage(), &id), + Some(CachedBatchType::MemoryArrow) + ); + } + + morsels.extend(next.plan().unwrap().unwrap().take_morsels()); + assert_eq!( + collect_columns(morsels).await.0, + vec![0, 1, 2, 3, 4, 5, 6, 7] + ); + } + + #[tokio::test] + async fn prefetched_multi_column_or_uses_snapshots() { + let predicate: Arc = Arc::new(BinaryExpr::new( + eq_expr("a", 0, 3), + Operator::Or, + eq_expr("b", 1, 20), + )); + assert!(extract_multi_column_or(&predicate).is_some()); + let file = prepare_test_file(PlanOptions { + predicate: Some(predicate), + ..Default::default() + }) + .await; + file.cache.storage().stats(); + + let chain = advance_to_row_group_chain(file.planner).await; + let morsels = chain.plan().unwrap().unwrap().take_morsels(); + assert_eq!(collect_columns(morsels).await, (vec![3], vec![13])); + assert_eq!( + file.cache.storage().stats().runtime.try_read_liquid_calls, + 0 + ); + } + + #[tokio::test] + async fn prefetch_fetches_absent_predicate_columns() { + let file = prepare_test_file(PlanOptions { + predicate: Some(gt_expr("a", 0, 7)), + projection_columns: vec![1], + single_row_group_values: Some((0..12).collect()), + ..Default::default() + }) + .await; + + let chain = advance_to_row_group_chain(file.planner).await; + let predicate = file + .cached_file + .create_row_group(0, vec![0]) + .get_column(0) + .unwrap(); + for batch_idx in 0..3 { + let entry_id = predicate.entry_id(BatchID::from_raw(batch_idx)).into(); + assert_eq!( + kind_of(file.cache.storage(), &entry_id), + Some(CachedBatchType::MemoryArrow) + ); + } + + let morsels = chain.plan().unwrap().unwrap().take_morsels(); + assert_eq!( + collect_columns(morsels).await.0, + vec![1008, 1009, 1010, 1011] + ); + } + + #[tokio::test] + async fn prefetch_skips_projection_for_filtered_batches() { + let file = prepare_test_file(PlanOptions { + predicate: Some(gt_expr("a", 0, 7)), + projection_columns: vec![1], + single_row_group_values: Some((0..12).collect()), + ..Default::default() + }) + .await; + let row_group = file.cached_file.create_row_group(0, vec![0]); + insert_batches( + &row_group, + 0, + &[(0, &[0, 1, 2, 3]), (1, &[4, 5, 6, 7]), (2, &[8, 9, 10, 11])], + ) + .await; + insert_batches( + &row_group, + 1, + &[ + (0, &[1000, 1001, 1002, 1003]), + (1, &[1004, 1005, 1006, 1007]), + (2, &[1008, 1009, 1010, 1011]), + ], + ) + .await; + file.cache.flush_data().await.unwrap(); + + let chain = advance_to_row_group_chain(file.planner).await; + let projection = row_group.get_column(1).unwrap(); + for batch_idx in 0..2 { + let entry_id = projection.entry_id(BatchID::from_raw(batch_idx)).into(); + assert_eq!( + kind_of(file.cache.storage(), &entry_id), + Some(CachedBatchType::DiskArrow) + ); + } + let surviving_entry = projection.entry_id(BatchID::from_raw(2)).into(); + assert_eq!( + kind_of(file.cache.storage(), &surviving_entry), + Some(CachedBatchType::MemoryArrow) + ); + + let morsels = chain.plan().unwrap().unwrap().take_morsels(); + assert_eq!( + collect_columns(morsels).await.0, + vec![1008, 1009, 1010, 1011] + ); + } + + #[tokio::test] + async fn snapshots_survive_eviction() { + let file = prepare_test_file(PlanOptions::default()).await; + let chain = advance_to_row_group_chain(file.planner).await; + let mut first = chain.plan().unwrap().unwrap(); + let next = first.take_pending_planner().unwrap().await.unwrap(); + let second = next.plan().unwrap().unwrap().take_morsels(); + file.cache.flush_data().await.unwrap(); + + let row_group = file.cached_file.create_row_group(1, vec![]); + assert_eq!(collect_columns(second).await.0, vec![4, 5, 6, 7]); + for column_id in 0..2 { + let id = row_group + .get_column(column_id) + .unwrap() + .entry_id(BatchID::from_raw(0)) + .into(); + assert_eq!( + kind_of(file.cache.storage(), &id), + Some(CachedBatchType::DiskArrow) + ); + } + } + + #[tokio::test] + async fn headroom_gate_skips_prefetch() { + let file = prepare_test_file(PlanOptions { + max_memory_bytes: 1, + max_disk_bytes: 0, + ..Default::default() + }) + .await; + let chain = advance_to_row_group_chain(file.planner).await; + let mut first = chain.plan().unwrap().unwrap(); + assert!(first.take_pending_planner().is_none()); + let next = first.take_ready_planners().pop().unwrap(); + + let mut morsels = first.take_morsels(); + morsels.extend(next.plan().unwrap().unwrap().take_morsels()); + assert_eq!( + collect_columns(morsels).await.0, + vec![0, 1, 2, 3, 4, 5, 6, 7] + ); + } + + #[tokio::test] + async fn cache_full_keeps_inserted_batches_and_skips_failed_inserts() { + let one_array_memory = Arc::new(Int32Array::from(vec![0, 1, 2, 3])).get_array_memory_size(); + let planned = plan_test_file(PlanOptions { + max_memory_bytes: one_array_memory * 3, + max_disk_bytes: 0, + ..Default::default() + }) + .await; + let row_group0 = planned.cached_file.create_row_group(0, vec![]); + let row_group1 = planned.cached_file.create_row_group(1, vec![]); + + let (a, b) = collect_columns(planned.morsels).await; + assert_eq!(a, vec![0, 1, 2, 3, 4, 5, 6, 7]); + assert_eq!(b, vec![10, 11, 12, 13, 14, 15, 16, 17]); + assert!(contains(&row_group0, 0, 0).await); + assert!(contains(&row_group0, 1, 0).await); + assert!(contains(&row_group1, 0, 0).await); + assert!(!contains(&row_group1, 1, 0).await); + } + + #[tokio::test] + async fn cache_full_with_filter_keeps_results_correct() { + let one_array_memory = Arc::new(Int32Array::from(vec![0, 1, 2, 3])).get_array_memory_size(); + let planned = plan_test_file(PlanOptions { + max_memory_bytes: one_array_memory * 3, + max_disk_bytes: 0, + predicate: Some(gt_expr("a", 0, 2)), + ..Default::default() + }) + .await; + let row_group0 = planned.cached_file.create_row_group(0, vec![]); + let row_group1 = planned.cached_file.create_row_group(1, vec![]); + let (a, b) = collect_columns(planned.morsels).await; + assert_eq!(a, vec![3, 4, 5, 6, 7]); + assert_eq!(b, vec![13, 14, 15, 16, 17]); + assert!(contains(&row_group0, 0, 0).await); + assert!(contains(&row_group0, 1, 0).await); + assert!(contains(&row_group1, 0, 0).await); + assert!(!contains(&row_group1, 1, 0).await); + } + + #[tokio::test] + async fn mid_scan_eviction_recovers() { + let planned = plan_test_file(PlanOptions { + max_memory_bytes: 0, + max_disk_bytes: 0, + ..Default::default() + }) + .await; + let row_group0 = planned.cached_file.create_row_group(0, vec![]); + let row_group1 = planned.cached_file.create_row_group(1, vec![]); + let (a, b) = collect_columns(planned.morsels).await; + assert_eq!(a, vec![0, 1, 2, 3, 4, 5, 6, 7]); + assert_eq!(b, vec![10, 11, 12, 13, 14, 15, 16, 17]); + for row_group in [&row_group0, &row_group1] { + assert!(!contains(row_group, 0, 0).await); + assert!(!contains(row_group, 1, 0).await); + } + } + + #[tokio::test] + async fn predicate_fallback_uses_predicate_projection() { + let one_array_memory = Arc::new(Int32Array::from(vec![0, 1, 2, 3])).get_array_memory_size(); + let planned = plan_test_file(PlanOptions { + max_memory_bytes: one_array_memory * 3, + max_disk_bytes: 0, + predicate: Some(gt_expr("b", 1, 12)), + projection_columns: vec![0], + ..Default::default() + }) + .await; + let row_group0 = planned.cached_file.create_row_group(0, vec![]); + let row_group1 = planned.cached_file.create_row_group(1, vec![]); + assert_eq!( + collect_columns(planned.morsels).await.0, + vec![3, 4, 5, 6, 7] + ); + assert!(contains(&row_group0, 0, 0).await); + assert!(contains(&row_group0, 1, 0).await); + assert!(contains(&row_group1, 0, 0).await); + assert!(!contains(&row_group1, 1, 0).await); + } + + #[tokio::test] + async fn missing_column_falls_back_to_parquet() { + let file = prepare_test_file(PlanOptions::default()).await; + let row_group0 = file.cached_file.create_row_group(0, vec![]); + let row_group1 = file.cached_file.create_row_group(1, vec![]); + insert_batches(&row_group0, 0, &[(0, &[0, 1, 2, 3])]).await; + insert_batches(&row_group1, 0, &[(0, &[4, 5, 6, 7])]).await; + + let (a, b) = collect_columns(drive_planner(file.planner).await).await; + assert_eq!(a, vec![0, 1, 2, 3, 4, 5, 6, 7]); + assert_eq!(b, vec![10, 11, 12, 13, 14, 15, 16, 17]); + assert!(contains(&row_group0, 1, 0).await); + assert!(contains(&row_group1, 1, 0).await); + } + + #[tokio::test] + async fn fallback_stream_advances_across_misses() { + let parquet_a = vec![ + 100, 101, 102, 103, 4, 5, 6, 7, 200, 201, 202, 203, 12, 13, 14, 15, + ]; + let file = prepare_test_file(PlanOptions { + projection_columns: vec![0], + single_row_group_values: Some(parquet_a), + ..Default::default() + }) + .await; + let row_group = file.cached_file.create_row_group(0, vec![]); + insert_batches(&row_group, 0, &[(0, &[0, 1, 2, 3]), (2, &[8, 9, 10, 11])]).await; + + assert_eq!( + collect_columns(drive_planner(file.planner).await).await.0, + (0..16).collect::>() + ); + for batch_idx in 0..4 { + assert!(contains(&row_group, 0, batch_idx).await); + } + } + + #[tokio::test] + async fn source_uses_native_morsel_api() { + let schema = schema(); + let tmp_dir = tempfile::tempdir().unwrap(); + let parquet_path = tmp_dir.path().join("data.parquet"); + write_two_row_group_file(&parquet_path, Arc::clone(&schema)); + let file = PartitionedFile::new( + "data.parquet", + std::fs::metadata(&parquet_path).unwrap().len(), + ); + let cache = create_test_cache(tmp_dir.path(), usize::MAX, usize::MAX).await; + let source = LiquidParquetSource::from_parquet_source( + ParquetSource::new(Arc::clone(&schema)), + cache, + ); + let base_config = FileScanConfigBuilder::new( + ObjectStoreUrl::local_filesystem(), + Arc::new(source.clone()), + ) + .with_file(file.clone()) + .build(); + let object_store = Arc::new(LocalFileSystem::new_with_prefix(tmp_dir.path()).unwrap()); + + assert!( + source + .create_file_opener(object_store.clone(), &base_config, 0) + .is_err() + ); + let morselizer = source + .create_morselizer(object_store, &base_config, 0) + .unwrap(); + assert!(morselizer.plan_file(file).is_ok()); + } +} diff --git a/src/datafusion/src/reader/plantime/opener.rs b/src/datafusion/src/reader/plantime/opener.rs deleted file mode 100644 index 7bc5e1b0c..000000000 --- a/src/datafusion/src/reader/plantime/opener.rs +++ /dev/null @@ -1,408 +0,0 @@ -use std::sync::Arc; - -use crate::{ - cache::{ColumnSqueezeHints, LiquidCacheParquetRef}, - reader::{ - plantime::{row_filter::build_row_filter, row_group_filter::RowGroupAccessPlanFilter}, - runtime::LiquidStreamBuilder, - }, -}; -use arrow::array::{RecordBatch, RecordBatchOptions}; -use arrow_schema::SchemaRef; -use datafusion::{ - common::exec_err, - datasource::{ - listing::PartitionedFile, - physical_plan::{ - FileOpenFuture, FileOpener, ParquetFileMetrics, - parquet::{PagePruningAccessPlanFilter, ParquetAccessPlan}, - }, - table_schema::TableSchema, - }, - error::DataFusionError, - physical_expr::PhysicalExprSimplifier, - physical_expr::projection::ProjectionExprs, - physical_expr::utils::reassign_expr_columns, - physical_expr_adapter::{PhysicalExprAdapterFactory, replace_columns_with_literals}, - physical_optimizer::pruning::{FilePruner, PruningPredicate, build_pruning_predicate}, - physical_plan::{ - PhysicalExpr, - metrics::{Count, ExecutionPlanMetricsSet, MetricBuilder}, - }, -}; -use futures::StreamExt; -use futures::TryStreamExt; -use parquet::arrow::{ - ParquetRecordBatchStreamBuilder, ProjectionMask, - arrow_reader::{ArrowReaderMetadata, ArrowReaderOptions}, -}; -use parquet::file::metadata::ParquetMetaData; - -use super::source::CachedMetaReaderFactory; - -pub struct LiquidParquetOpener { - partition_index: usize, - projection: ProjectionExprs, - limit: Option, - predicate: Option>, - table_schema: TableSchema, - metrics: ExecutionPlanMetricsSet, - parquet_file_reader_factory: Arc, - reorder_filters: bool, - liquid_cache: LiquidCacheParquetRef, - expr_adapter_factory: Arc, - span: Option>, - squeeze_hints: Arc, -} - -impl LiquidParquetOpener { - #[allow(clippy::too_many_arguments)] - pub fn new( - partition_index: usize, - projection: ProjectionExprs, - limit: Option, - predicate: Option>, - table_schema: TableSchema, - metrics: ExecutionPlanMetricsSet, - liquid_cache: LiquidCacheParquetRef, - parquet_file_reader_factory: Arc, - reorder_filters: bool, - expr_adapter_factory: Arc, - span: Option>, - squeeze_hints: Arc, - ) -> Self { - Self { - partition_index, - projection, - limit, - predicate, - table_schema, - metrics, - liquid_cache, - parquet_file_reader_factory, - reorder_filters, - expr_adapter_factory, - span, - squeeze_hints, - } - } -} - -impl FileOpener for LiquidParquetOpener { - fn open(&self, partitioned_file: PartitionedFile) -> Result { - let file_range = partitioned_file.range.clone(); - let access_plan_ext = partitioned_file.extensions.get_arc::(); - let file_name = partitioned_file.object_meta.location.to_string(); - let file_metrics = ParquetFileMetrics::new(self.partition_index, &file_name, &self.metrics); - - let metadata_size_hint = partitioned_file.metadata_size_hint; - - let lc = self.liquid_cache.clone(); - let file_loc = partitioned_file.object_meta.location.to_string(); - - let mut async_file_reader = self.parquet_file_reader_factory.create_liquid_reader( - self.partition_index, - partitioned_file.clone(), - metadata_size_hint, - &self.metrics, - ); - - let logical_file_schema = Arc::clone(self.table_schema.file_schema()); - let output_schema = Arc::new( - self.projection - .project_schema(self.table_schema.table_schema())?, - ); - let mut projection = self.projection.clone(); - let mut predicate = self.predicate.clone(); - let mut literal_columns = std::collections::HashMap::new(); - for (field, value) in self - .table_schema - .table_partition_cols() - .iter() - .zip(partitioned_file.partition_values.iter()) - { - literal_columns.insert(field.name().clone(), value.clone()); - } - if !literal_columns.is_empty() { - projection = projection.try_map_exprs(|expr| { - replace_columns_with_literals(Arc::clone(&expr), &literal_columns) - })?; - predicate = predicate - .map(|p| replace_columns_with_literals(p, &literal_columns)) - .transpose()?; - } - let reorder_predicates = self.reorder_filters; - let limit = self.limit; - - let predicate_creation_errors = - MetricBuilder::new(&self.metrics).global_counter("num_predicate_creation_errors"); - - let expr_adapter_factory = Arc::clone(&self.expr_adapter_factory); - let span = self.span.clone(); - let squeeze_hints = Arc::clone(&self.squeeze_hints); - Ok(Box::pin(async move { - // Prune this file using the file level statistics and partition values. - // Since dynamic filters may have been updated since planning it is possible that we are able - // to prune files now that we couldn't prune at planning time. - // We'll also check this after every record batch we read, - // and if at some point we are able to prove we can prune the file using just the file level statistics - // we can end the stream early. - // `FilePruner::try_new` itself decides whether a pruner is worth - // building: it returns `None` for a purely static predicate over a - // file with no usable column statistics. - let mut file_pruner = predicate.as_ref().and_then(|p| { - FilePruner::try_new( - Arc::clone(p), - &logical_file_schema, - &partitioned_file, - predicate_creation_errors.clone(), - ) - }); - - if let Some(file_pruner) = &mut file_pruner - && file_pruner.should_prune()? - { - file_metrics.files_ranges_pruned_statistics.add_pruned(1); - return Ok(futures::stream::empty().boxed()); - } - - file_metrics.files_ranges_pruned_statistics.add_matched(1); - - // `Optional`, not `Required`: the page index is an optimization, not - // a correctness requirement. It drives page-level pruning below, - // which no-ops when the index is absent. `Required` instead fails - // the whole read with `missing offset index` on any file that - // advertises a page index while one of its column chunks carries no - // offset index — a shape valid parquet is free to have. - let mut options = ArrowReaderOptions::new() - .with_page_index_policy(parquet::file::metadata::PageIndexPolicy::Optional); - let mut metadata_timer = file_metrics.metadata_load_time.timer(); - - // Begin by loading the metadata from the underlying reader (note - // the returned metadata may actually include page indexes as some - // readers may return page indexes even when not requested -- for - // example when they are cached) - let mut reader_metadata = - ArrowReaderMetadata::load_async(&mut async_file_reader, options.clone()).await?; - - // Note about schemas: we are actually dealing with **3 different schemas** here: - // - The table schema as defined by the TableProvider. - // This is what the user sees, what they get when they `SELECT * FROM table`, etc. - // - The logical file schema: this is the table schema minus any hive partition columns and projections. - // This is what the physical file schema is coerced to. - // - The physical file schema: this is the schema as defined by the parquet file. This is what the parquet file actually contains. - let physical_file_schema = Arc::clone(reader_metadata.schema()); - let cache_full_schema = Arc::clone(&physical_file_schema); - options = options.with_schema(Arc::clone(&physical_file_schema)); - reader_metadata = - ArrowReaderMetadata::try_new(Arc::clone(reader_metadata.metadata()), options)?; - debug_assert!( - Arc::strong_count(reader_metadata.metadata()) > 1, - "meta data must be cached already" - ); - - let rewriter = expr_adapter_factory.create( - Arc::clone(&logical_file_schema), - Arc::clone(&physical_file_schema), - )?; - let simplifier = PhysicalExprSimplifier::new(&physical_file_schema); - predicate = predicate - .map(|p| simplifier.simplify(rewriter.rewrite(p)?)) - .transpose()?; - projection = projection.try_map_exprs(|p| simplifier.simplify(rewriter.rewrite(p)?))?; - - let (pruning_predicate, page_pruning_predicate) = build_pruning_predicates( - predicate.as_ref(), - &physical_file_schema, - &predicate_creation_errors, - ); - - metadata_timer.stop(); - - let mut builder = ParquetRecordBatchStreamBuilder::new_with_metadata( - async_file_reader.clone(), - reader_metadata.clone(), - ); - let indices = projection.column_indices(); - let mask = ProjectionMask::roots(builder.parquet_schema(), indices); - - // Filter pushdown: evaluate predicates during scan. - // - // A failure here is not recoverable by ignoring it. DataFusion removed - // the `FilterExec` when it pushed this predicate down, so the row - // filter is the only place the predicate is applied; carrying on - // without one returns rows the query excluded. Fail the query instead - // (issue #23). - let row_filter = match predicate.as_ref() { - Some(p) => build_row_filter( - p, - &physical_file_schema, - reader_metadata.metadata(), - reorder_predicates, - &file_metrics, - )?, - None => None, - }; - - // Determine which row groups to actually read. The idea is to skip - // as many row groups as possible based on the metadata and query - let file_metadata: Arc = Arc::clone(builder.metadata()); - let predicate = pruning_predicate.as_ref().map(|p| p.as_ref()); - let rg_metadata = file_metadata.row_groups(); - // track which row groups to actually read - let access_plan = create_initial_plan(&file_name, access_plan_ext, rg_metadata.len())?; - let mut row_groups = RowGroupAccessPlanFilter::new(access_plan); - // if there is a range restricting what parts of the file to read - if let Some(range) = file_range.as_ref() { - row_groups.prune_by_range(rg_metadata, range); - } - // If there is a predicate that can be evaluated against the metadata - if let Some(predicate) = predicate.as_ref() { - row_groups.prune_by_statistics( - &physical_file_schema, - builder.parquet_schema(), - rg_metadata, - predicate, - &file_metrics, - ); - - if !row_groups.is_empty() { - row_groups - .prune_by_bloom_filters( - &physical_file_schema, - &mut builder, - predicate, - &file_metrics, - ) - .await; - } - } - - let mut access_plan = row_groups.build(); - - // page index pruning: if all data on individual pages can - // be ruled using page metadata, rows from other columns - // with that range can be skipped as well - if !access_plan.is_empty() - && let Some(p) = page_pruning_predicate - { - access_plan = p.prune_plan_with_page_index( - access_plan, - &physical_file_schema, - builder.parquet_schema(), - file_metadata.as_ref(), - &file_metrics, - ); - } - - let row_group_indexes = access_plan.row_group_indexes(); - let row_selection = access_plan.into_overall_row_selection(rg_metadata)?; - - let mut liquid_builder = - LiquidStreamBuilder::new(async_file_reader, Arc::clone(reader_metadata.metadata())) - .with_row_groups(row_group_indexes) - .with_projection(mask) - .with_selection(row_selection) - .with_limit(limit); - - if let Some(row_filter) = row_filter { - liquid_builder = liquid_builder.with_row_filter(row_filter); - } - - if let Some(s) = &span { - let span = fastrace::Span::enter_with_parent("liquid_stream", s); - liquid_builder = liquid_builder.with_span(span); - } - - let liquid_cache = lc.register_or_get_file_with_hints( - file_loc, - Arc::clone(&cache_full_schema), - squeeze_hints, - ); - - let stream = liquid_builder.build(liquid_cache)?; - - let stream_schema = Arc::clone(stream.schema()); - let replace_schema = !stream_schema.eq(&output_schema); - let projection = - projection.try_map_exprs(|expr| reassign_expr_columns(expr, &stream_schema))?; - let projector = projection.make_projector(&stream_schema)?; - - let adapted = stream - .map_err(|e| DataFusionError::External(Box::new(e))) - .map(move |batch| { - batch.and_then(|batch| { - let batch = projector.project_batch(&batch)?; - if replace_schema { - let (_schema, arrays, num_rows) = batch.into_parts(); - let options = RecordBatchOptions::new().with_row_count(Some(num_rows)); - RecordBatch::try_new_with_options( - Arc::clone(&output_schema), - arrays, - &options, - ) - .map_err(Into::into) - } else { - Ok(batch) - } - }) - }); - - Ok(adapted.boxed()) - })) - } -} - -fn create_initial_plan( - file_name: &str, - access_plan: Option>, - row_group_count: usize, -) -> Result { - if let Some(access_plan) = access_plan { - let plan_len = access_plan.len(); - if plan_len != row_group_count { - return exec_err!( - "Invalid ParquetAccessPlan for {file_name}. Specified {plan_len} row groups, but file has {row_group_count}" - ); - } - - // check row group count matches the plan - return Ok(access_plan.as_ref().clone()); - } - - // default to scanning all row groups - Ok(ParquetAccessPlan::new_all(row_group_count)) -} - -pub(crate) fn build_pruning_predicates( - predicate: Option<&Arc>, - file_schema: &SchemaRef, - predicate_creation_errors: &Count, -) -> ( - Option>, - Option>, -) { - let Some(predicate) = predicate.as_ref() else { - return (None, None); - }; - let pruning_predicate = build_pruning_predicate( - Arc::clone(predicate), - file_schema, - predicate_creation_errors, - ); - let page_pruning_predicate = build_page_pruning_predicate(predicate, file_schema); - (pruning_predicate, Some(page_pruning_predicate)) -} - -/// Build a page pruning predicate from an optional predicate expression. -/// If the predicate is None or the predicate cannot be converted to a page pruning -/// predicate, return None. -pub(crate) fn build_page_pruning_predicate( - predicate: &Arc, - file_schema: &SchemaRef, -) -> Arc { - Arc::new(PagePruningAccessPlanFilter::new( - predicate, - Arc::clone(file_schema), - )) -} diff --git a/src/datafusion/src/reader/plantime/row_filter.rs b/src/datafusion/src/reader/plantime/row_filter.rs index 1eb16c32b..e4efe3b43 100644 --- a/src/datafusion/src/reader/plantime/row_filter.rs +++ b/src/datafusion/src/reader/plantime/row_filter.rs @@ -64,7 +64,7 @@ use std::collections::BTreeSet; use std::sync::Arc; use arrow::array::BooleanArray; -use arrow::datatypes::{DataType, Schema}; +use arrow::datatypes::Schema; use arrow::error::{ArrowError, Result as ArrowResult}; use arrow::record_batch::RecordBatch; use arrow_schema::SchemaRef; @@ -85,6 +85,7 @@ use datafusion::physical_expr::expressions::Column; use datafusion::physical_expr::{PhysicalExpr, split_conjunction}; /// A row filter that can be used to filter rows from a parquet file. +#[derive(Clone)] pub struct LiquidRowFilter { predicates: Vec, } @@ -106,24 +107,6 @@ impl LiquidRowFilter { } } -pub(crate) fn get_predicate_column_id(projection: &parquet::arrow::ProjectionMask) -> Vec { - #[derive(Debug, Clone)] - struct ProjectionMaskLiquid { - mask: Option>, - } - let project_inner: &ProjectionMaskLiquid = unsafe { std::mem::transmute(projection) }; - project_inner - .mask - .as_ref() - .map(|m| { - m.iter() - .enumerate() - .filter_map(|(pos, &x)| if x { Some(pos) } else { None }) - .collect::>() - }) - .unwrap_or_default() -} - /// A "compiled" predicate passed to `ParquetRecordBatchStream` to perform /// row-level filtering during parquet decoding. /// @@ -133,7 +116,7 @@ pub(crate) fn get_predicate_column_id(projection: &parquet::arrow::ProjectionMas /// /// An expression can be evaluated as a `DatafusionArrowPredicate` if it: /// * Does not reference any projected columns -/// * Does not reference columns with non-primitive types (e.g. structs / lists) +/// * References only columns present in the physical file schema #[derive(Debug, Clone)] pub struct LiquidPredicate { /// the filter expression @@ -143,6 +126,9 @@ pub struct LiquidPredicate { /// Path to the columns in the parquet schema required to evaluate the /// expression projection_mask: ProjectionMask, + /// Indices into the file schema of the columns required to evaluate the + /// expression, in `filter_schema` order + column_ids: Vec, /// how many rows were filtered out by this predicate rows_pruned: metrics::Count, /// how many rows passed this predicate @@ -173,6 +159,7 @@ impl LiquidPredicate { physical_expr, physical_expr_physical_column_index: candidate.expr, projection_mask: projection, + column_ids: candidate.projection, rows_pruned, rows_matched, time, @@ -202,8 +189,7 @@ impl LiquidPredicate { /// Get the column ids of the predicate. pub fn predicate_column_ids(&self) -> Vec { - let projection = self.projection(); - get_predicate_column_id(projection) + self.column_ids.clone() } } @@ -313,11 +299,9 @@ impl FilterCandidateBuilder { // a struct that implements TreeNodeRewriter to traverse a PhysicalExpr tree structure to determine // if any column references in the expression would prevent it from being predicate-pushed-down. -// if non_primitive_columns || projected_columns, it can't be pushed down. +// if projected_columns, it can't be pushed down. // can't be reused between calls to `rewrite`; each construction must be used only once. struct PushdownChecker<'schema> { - /// Does the expression require any non-primitive columns (like structs)? - non_primitive_columns: bool, /// Does the expression reference any columns that are not in the file schema? projected_columns: bool, // Indices into the file schema of the columns required to evaluate the expression @@ -328,7 +312,6 @@ struct PushdownChecker<'schema> { impl<'schema> PushdownChecker<'schema> { fn new(file_schema: &'schema Schema) -> Self { Self { - non_primitive_columns: false, projected_columns: false, required_columns: BTreeSet::default(), file_schema, @@ -338,10 +321,6 @@ impl<'schema> PushdownChecker<'schema> { fn check_single_column(&mut self, column_name: &str) -> Option { if let Ok(idx) = self.file_schema.index_of(column_name) { self.required_columns.insert(idx); - if DataType::is_nested(self.file_schema.field(idx).data_type()) { - self.non_primitive_columns = true; - return Some(TreeNodeRecursion::Jump); - } } else { // If the column does not exist in the file schema then it cannot be pushed down. self.projected_columns = true; @@ -353,7 +332,7 @@ impl<'schema> PushdownChecker<'schema> { #[inline] fn prevents_pushdown(&self) -> bool { - self.non_primitive_columns || self.projected_columns + self.projected_columns } } @@ -399,8 +378,9 @@ fn pushdown_columns( /// schema but are turned into literals by the opener's physical-expr adapter /// before the row filter ever sees them — refusing on their account would /// needlessly bypass the cache. What genuinely cannot be evaluated is a -/// reference to a nested column, or to a column that exists nowhere in the -/// table. +/// reference to a column that exists nowhere in the table. (A nested column is +/// fine: the cache holds what it cannot transcode as Arrow and the predicate +/// evaluates against that.) pub fn unevaluable_conjunct<'e>( expr: &'e Arc, schema: &Schema, @@ -578,7 +558,7 @@ fn get_priority(expr: &Arc) -> u8 { mod tests { use super::*; use arrow::array::{Int32Array, Int64Array, RecordBatch, StructArray}; - use arrow_schema::{Field, Fields}; + use arrow_schema::{DataType, Field, Fields}; use datafusion::common::ScalarValue; use datafusion::physical_plan::expressions::{BinaryExpr, Column, Literal}; use datafusion::physical_plan::metrics::ExecutionPlanMetricsSet; @@ -642,16 +622,19 @@ mod tests { (schema, metadata) } - /// `PushdownChecker`'s `non_primitive_columns` path: `st` is a struct, so no - /// conjunct that reads it can be evaluated as an `ArrowPredicate`. + /// A nested column is evaluable: `PushdownChecker` no longer refuses structs. + /// The blanket "non-primitive" bar was inherited from DataFusion's own + /// `row_filter.rs` and says nothing about what LiquidCache can do — a column + /// it cannot transcode is simply held as Arrow, in memory or demoted to disk + /// under squeeze pressure, and the predicate evaluates against that. Declining + /// instead cost the whole scan its cache for any table carrying one struct + /// column. #[test] - fn nested_column_conjunct_is_unevaluable() { + fn nested_column_conjunct_is_evaluable() { let schema = schema_with_struct(); - let nested = eq(col("st", 1), 3); - let expr = and(eq(col("id", 0), 0), Arc::clone(&nested)); + let expr = and(eq(col("id", 0), 0), eq(col("st", 1), 3)); - let found = unevaluable_conjunct(&expr, &schema).unwrap(); - assert!(Arc::ptr_eq(found.unwrap(), &nested)); + assert!(unevaluable_conjunct(&expr, &schema).unwrap().is_none()); } /// `PushdownChecker`'s `projected_columns` path: `missing` is in no schema the @@ -683,7 +666,9 @@ mod tests { #[cfg_attr(debug_assertions, should_panic(expected = "row filter dropped"))] fn build_row_filter_refuses_to_drop_a_conjunct() { let (schema, metadata) = metadata(); - let expr = and(eq(col("id", 0), 0), eq(col("st", 1), 3)); + // A column in no schema the scan can read -- the one path that still + // makes a conjunct unevaluable now that nested columns push down. + let expr = and(eq(col("id", 0), 0), eq(col("missing", 2), 3)); let metrics = ExecutionPlanMetricsSet::new(); let file_metrics = ParquetFileMetrics::new(0, "test.parquet", &metrics); diff --git a/src/datafusion/src/reader/plantime/row_group_filter.rs b/src/datafusion/src/reader/plantime/row_group_filter.rs deleted file mode 100644 index eb26b34b0..000000000 --- a/src/datafusion/src/reader/plantime/row_group_filter.rs +++ /dev/null @@ -1,427 +0,0 @@ -// Licensed to the Apache Software Foundation (ASF) under one -// or more contributor license agreements. See the NOTICE file -// distributed with this work for additional information -// regarding copyright ownership. The ASF licenses this file -// to you under the Apache License, Version 2.0 (the -// "License"); you may not use this file except in compliance -// with the License. You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, -// software distributed under the License is distributed on an -// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY -// KIND, either express or implied. See the License for the -// specific language governing permissions and limitations -// under the License. - -use arrow::{array::ArrayRef, array::BooleanArray, array::UInt64Array, datatypes::Schema}; -use datafusion::common::{Column, Result, ScalarValue}; -use datafusion::datasource::listing::FileRange; -use datafusion::datasource::physical_plan::ParquetFileMetrics; -use datafusion::datasource::physical_plan::parquet::ParquetAccessPlan; -use datafusion::physical_optimizer::pruning::{PruningPredicate, PruningStatistics}; -use parquet::arrow::arrow_reader::statistics::StatisticsConverter; -use parquet::arrow::parquet_column; -use parquet::basic::Type; -use parquet::data_type::Decimal; -use parquet::schema::types::SchemaDescriptor; -use parquet::{ - arrow::{ParquetRecordBatchStreamBuilder, async_reader::AsyncFileReader}, - bloom_filter::Sbbf, - file::metadata::RowGroupMetaData, -}; -use std::collections::{HashMap, HashSet}; -use std::sync::Arc; - -/// Reduces the [`ParquetAccessPlan`] based on row group level metadata. -/// -/// This struct implements the various types of pruning that are applied to a -/// set of row groups within a parquet file, progressively narrowing down the -/// set of row groups (and ranges/selections within those row groups) that -/// should be scanned, based on the available metadata. -#[derive(Debug, Clone, PartialEq)] -pub struct RowGroupAccessPlanFilter { - /// which row groups should be accessed - access_plan: ParquetAccessPlan, -} - -impl RowGroupAccessPlanFilter { - /// Create a new `RowGroupPlanBuilder` for pruning out the groups to scan - /// based on metadata and statistics - pub fn new(access_plan: ParquetAccessPlan) -> Self { - Self { access_plan } - } - - /// Return true if there are no row groups - pub fn is_empty(&self) -> bool { - self.access_plan.is_empty() - } - - /// Returns the inner access plan - pub fn build(self) -> ParquetAccessPlan { - self.access_plan - } - - /// Prune remaining row groups to only those within the specified range. - /// - /// Updates this set to mark row groups that should not be scanned - /// - /// # Panics - /// if `groups.len() != self.len()` - pub fn prune_by_range(&mut self, groups: &[RowGroupMetaData], range: &FileRange) { - assert_eq!(groups.len(), self.access_plan.len()); - for (idx, metadata) in groups.iter().enumerate() { - if !self.access_plan.should_scan(idx) { - continue; - } - - // Skip the row group if the first dictionary/data page are not - // within the range. - // - // note don't use the location of metadata - // - let col = metadata.column(0); - let offset = col - .dictionary_page_offset() - .unwrap_or_else(|| col.data_page_offset()); - if !range.contains(offset) { - self.access_plan.skip(idx); - } - } - } - /// Prune remaining row groups using min/max/null_count statistics and - /// the [`PruningPredicate`] to determine if the predicate can not be true. - /// - /// Updates this set to mark row groups that should not be scanned - /// - /// Note: This method currently ignores ColumnOrder - /// - /// - /// # Panics - /// if `groups.len() != self.len()` - pub fn prune_by_statistics( - &mut self, - arrow_schema: &Schema, - parquet_schema: &SchemaDescriptor, - groups: &[RowGroupMetaData], - predicate: &PruningPredicate, - metrics: &ParquetFileMetrics, - ) { - // scoped timer updates on drop - let _timer_guard = metrics.statistics_eval_time.timer(); - - assert_eq!(groups.len(), self.access_plan.len()); - // Indexes of row groups still to scan - let row_group_indexes = self.access_plan.row_group_indexes(); - let row_group_metadatas = row_group_indexes - .iter() - .map(|&i| &groups[i]) - .collect::>(); - - let pruning_stats = RowGroupPruningStatistics { - parquet_schema, - row_group_metadatas, - arrow_schema, - }; - - // try to prune the row groups in a single call - match predicate.prune(&pruning_stats) { - Ok(values) => { - // values[i] is false means the predicate could not be true for row group i - for (idx, &value) in row_group_indexes.iter().zip(values.iter()) { - if !value { - self.access_plan.skip(*idx); - metrics.row_groups_pruned_statistics.add_pruned(1); - } else { - metrics.row_groups_pruned_statistics.add_matched(1); - } - } - } - // stats filter array could not be built, so we can't prune - Err(e) => { - log::debug!("Error evaluating row group predicate values {e}"); - metrics.predicate_evaluation_errors.add(1); - } - } - } - - /// Prune remaining row groups using available bloom filters and the - /// [`PruningPredicate`]. - /// - /// Updates this set with row groups that should not be scanned - /// - /// # Panics - /// if the builder does not have the same number of row groups as this set - pub async fn prune_by_bloom_filters( - &mut self, - arrow_schema: &Schema, - builder: &mut ParquetRecordBatchStreamBuilder, - predicate: &PruningPredicate, - metrics: &ParquetFileMetrics, - ) { - // scoped timer updates on drop - let _timer_guard = metrics.bloom_filter_eval_time.timer(); - - assert_eq!(builder.metadata().num_row_groups(), self.access_plan.len()); - for idx in 0..self.access_plan.len() { - if !self.access_plan.should_scan(idx) { - continue; - } - - // Attempt to find bloom filters for filtering this row group - let literal_columns = predicate.literal_columns(); - let mut column_sbbf = HashMap::with_capacity(literal_columns.len()); - - for column_name in literal_columns { - let Some((column_idx, _field)) = - parquet_column(builder.parquet_schema(), arrow_schema, &column_name) - else { - continue; - }; - - let bf = match builder - .get_row_group_column_bloom_filter(idx, column_idx) - .await - { - Ok(Some(bf)) => bf, - Ok(None) => continue, // no bloom filter for this column - Err(e) => { - log::debug!("Ignoring error reading bloom filter: {e}"); - metrics.predicate_evaluation_errors.add(1); - continue; - } - }; - let physical_type = builder.parquet_schema().column(column_idx).physical_type(); - - column_sbbf.insert(column_name.to_string(), (bf, physical_type)); - } - - let stats = BloomFilterStatistics { column_sbbf }; - - // Can this group be pruned? - let prune_group = match predicate.prune(&stats) { - Ok(values) => !values[0], - Err(e) => { - log::debug!("Error evaluating row group predicate on bloom filter: {e}"); - metrics.predicate_evaluation_errors.add(1); - false - } - }; - - if prune_group { - metrics.row_groups_pruned_bloom_filter.add_pruned(1); - self.access_plan.skip(idx) - } else if !stats.column_sbbf.is_empty() { - metrics.row_groups_pruned_bloom_filter.add_matched(1); - } - } - } -} -/// Implements [`PruningStatistics`] for Parquet Split Block Bloom Filters (SBBF) -struct BloomFilterStatistics { - /// Maps column name to the parquet bloom filter and parquet physical type - column_sbbf: HashMap, -} - -impl BloomFilterStatistics { - /// Helper function for checking if [`Sbbf`] filter contains [`ScalarValue`]. - /// - /// In case the type of scalar is not supported, returns `true`, assuming that the - /// value may be present. - fn check_scalar(sbbf: &Sbbf, value: &ScalarValue, parquet_type: &Type) -> bool { - match value { - ScalarValue::Utf8(Some(v)) - | ScalarValue::Utf8View(Some(v)) - | ScalarValue::LargeUtf8(Some(v)) => sbbf.check(&v.as_str()), - ScalarValue::Binary(Some(v)) - | ScalarValue::BinaryView(Some(v)) - | ScalarValue::LargeBinary(Some(v)) => sbbf.check(v), - ScalarValue::FixedSizeBinary(_size, Some(v)) => sbbf.check(v), - ScalarValue::Boolean(Some(v)) => sbbf.check(v), - ScalarValue::Float64(Some(v)) => sbbf.check(v), - ScalarValue::Float32(Some(v)) => sbbf.check(v), - ScalarValue::Int64(Some(v)) => sbbf.check(v), - ScalarValue::Int32(Some(v)) => sbbf.check(v), - ScalarValue::UInt64(Some(v)) => sbbf.check(v), - ScalarValue::UInt32(Some(v)) => sbbf.check(v), - ScalarValue::Decimal128(Some(v), p, s) => match parquet_type { - Type::INT32 => { - //https://github.com/apache/parquet-format/blob/eb4b31c1d64a01088d02a2f9aefc6c17c54cc6fc/Encodings.md?plain=1#L35-L42 - // All physical type are little-endian - if *p > 9 { - //DECIMAL can be used to annotate the following types: - // - // int32: for 1 <= precision <= 9 - // int64: for 1 <= precision <= 18 - return true; - } - let b = (*v as i32).to_le_bytes(); - // Use Decimal constructor after https://github.com/apache/arrow-rs/issues/5325 - let decimal = Decimal::Int32 { - value: b, - precision: *p as i32, - scale: *s as i32, - }; - sbbf.check(&decimal) - } - Type::INT64 => { - if *p > 18 { - return true; - } - let b = (*v as i64).to_le_bytes(); - let decimal = Decimal::Int64 { - value: b, - precision: *p as i32, - scale: *s as i32, - }; - sbbf.check(&decimal) - } - Type::FIXED_LEN_BYTE_ARRAY => { - // keep with from_bytes_to_i128 - let b = v.to_be_bytes().to_vec(); - // Use Decimal constructor after https://github.com/apache/arrow-rs/issues/5325 - let decimal = Decimal::Bytes { - value: b.into(), - precision: *p as i32, - scale: *s as i32, - }; - sbbf.check(&decimal) - } - _ => true, - }, - // One more pattern matching since not all data types are supported - // inside of a Dictionary - ScalarValue::Dictionary(_, inner) => match inner.as_ref() { - ScalarValue::Int32(_) - | ScalarValue::Int64(_) - | ScalarValue::UInt32(_) - | ScalarValue::UInt64(_) - | ScalarValue::Float32(_) - | ScalarValue::Float64(_) - | ScalarValue::Utf8(_) - | ScalarValue::LargeUtf8(_) - | ScalarValue::Binary(_) - | ScalarValue::LargeBinary(_) => { - BloomFilterStatistics::check_scalar(sbbf, inner, parquet_type) - } - _ => true, - }, - _ => true, - } - } -} - -impl PruningStatistics for BloomFilterStatistics { - fn min_values(&self, _column: &Column) -> Option { - None - } - - fn max_values(&self, _column: &Column) -> Option { - None - } - - fn num_containers(&self) -> usize { - 1 - } - - fn null_counts(&self, _column: &Column) -> Option { - None - } - - fn row_counts(&self) -> Option { - None - } - - /// Use bloom filters to determine if we are sure this column can not - /// possibly contain `values` - /// - /// The `contained` API returns false if the bloom filters knows that *ALL* - /// of the values in a column are not present. - fn contained(&self, column: &Column, values: &HashSet) -> Option { - let (sbbf, parquet_type) = self.column_sbbf.get(column.name.as_str())?; - - // Bloom filters are probabilistic data structures that can return false - // positives (i.e. it might return true even if the value is not - // present) however, the bloom filter will return `false` if the value is - // definitely not present. - - let known_not_present = values - .iter() - .map(|value| BloomFilterStatistics::check_scalar(sbbf, value, parquet_type)) - // The row group doesn't contain any of the values if - // all the checks are false - .all(|v| !v); - - let contains = if known_not_present { - Some(false) - } else { - // Given the bloom filter is probabilistic, we can't be sure that - // the row group actually contains the values. Return `None` to - // indicate this uncertainty - None - }; - - Some(BooleanArray::from(vec![contains])) - } -} - -/// Wraps a slice of [`RowGroupMetaData`] in a way that implements [`PruningStatistics`] -struct RowGroupPruningStatistics<'a> { - parquet_schema: &'a SchemaDescriptor, - row_group_metadatas: Vec<&'a RowGroupMetaData>, - arrow_schema: &'a Schema, -} - -impl<'a> RowGroupPruningStatistics<'a> { - /// Return an iterator over the row group metadata - fn metadata_iter(&'a self) -> impl Iterator + 'a { - self.row_group_metadatas.iter().copied() - } - - fn statistics_converter<'b>(&'a self, column: &'b Column) -> Result> { - Ok(StatisticsConverter::try_new( - &column.name, - self.arrow_schema, - self.parquet_schema, - )?) - } -} - -impl PruningStatistics for RowGroupPruningStatistics<'_> { - fn min_values(&self, column: &Column) -> Option { - self.statistics_converter(column) - .and_then(|c| Ok(c.row_group_mins(self.metadata_iter())?)) - .ok() - } - - fn max_values(&self, column: &Column) -> Option { - self.statistics_converter(column) - .and_then(|c| Ok(c.row_group_maxes(self.metadata_iter())?)) - .ok() - } - - fn num_containers(&self) -> usize { - self.row_group_metadatas.len() - } - - fn null_counts(&self, column: &Column) -> Option { - self.statistics_converter(column) - .and_then(|c| Ok(c.row_group_null_counts(self.metadata_iter())?)) - .ok() - .map(|counts| Arc::new(counts) as ArrayRef) - } - - fn row_counts(&self) -> Option { - // Row counts are container-level — read directly from row group metadata. - let counts: UInt64Array = self - .metadata_iter() - .map(|rg| Some(rg.num_rows() as u64)) - .collect(); - Some(Arc::new(counts) as ArrayRef) - } - - fn contained(&self, _column: &Column, _values: &HashSet) -> Option { - None - } -} diff --git a/src/datafusion/src/reader/plantime/source.rs b/src/datafusion/src/reader/plantime/source.rs index b9af3c9e2..8ac4be639 100644 --- a/src/datafusion/src/reader/plantime/source.rs +++ b/src/datafusion/src/reader/plantime/source.rs @@ -1,42 +1,39 @@ -use super::opener::LiquidParquetOpener; +use super::LiquidMorselizer; use crate::cache::{ColumnSqueezeHints, LiquidCacheParquetRef}; use ahash::{HashMap, HashMapExt}; -use arrow_schema::Schema; use bytes::Bytes; use datafusion::{ - common::tree_node::TreeNodeRecursion, - config::TableParquetOptions, + common::{internal_err, tree_node::TreeNodeRecursion}, + config::{ConfigOptions, TableParquetOptions}, datasource::{ listing::PartitionedFile, physical_plan::{ FileScanConfig, FileSource, ParquetFileMetrics, ParquetFileReaderFactory, - ParquetSource, parquet::PagePruningAccessPlanFilter, + ParquetSource, parquet::can_expr_be_pushed_down_with_schemas, }, table_schema::TableSchema, }, error::Result, + execution::object_store::ObjectStoreUrl, physical_expr::projection::ProjectionExprs, + physical_expr::utils::conjunction, physical_expr_adapter::DefaultPhysicalExprAdapterFactory, - physical_optimizer::pruning::{PruningPredicate, PruningPredicateBuilder}, physical_plan::{ - PhysicalExpr, apply_expression_roots, - metrics::{ExecutionPlanMetricsSet, MetricBuilder}, + DisplayFormatType, PhysicalExpr, apply_expression_roots, + filter_pushdown::{FilterPushdownPropagation, PushedDown, PushedDownPredicate}, + metrics::ExecutionPlanMetricsSet, }, }; +use datafusion_datasource::morsel::Morselizer; use futures::{FutureExt, future::BoxFuture}; -use object_store::{ObjectStore, path::Path}; -use parquet::arrow::arrow_reader::ArrowReaderOptions; -use parquet::arrow::async_reader::AsyncFileReader; -// `ParquetObjectReader` is deprecated in arrow-rs 59 in favour of implementing -// `AsyncFileReader` against the object store directly -// (https://github.com/apache/arrow-rs/issues/10308). We keep it for now: it is the -// only readily available reader that coalesces byte ranges, and DataFusion's -// `ParquetFileReader` — the suggested replacement — has a `pub(crate)` -// constructor and does no coalescing. -#[allow(deprecated)] -use parquet::arrow::async_reader::ParquetObjectReader; -use parquet::file::metadata::{PageIndexPolicy, ParquetMetaData, ParquetMetaDataReader}; +use object_store::{ObjectStore, ObjectStoreExt, path::Path}; +use parquet::{ + arrow::{arrow_reader::ArrowReaderOptions, async_reader::AsyncFileReader}, + errors::ParquetError, + file::metadata::{PageIndexPolicy, ParquetMetaData, ParquetMetaDataReader}, +}; use std::{ + fmt::{self, Formatter}, ops::Range, sync::{Arc, LazyLock}, }; @@ -47,11 +44,16 @@ static META_CACHE: LazyLock = LazyLock::new(MetadataCache::new); #[derive(Debug)] pub(crate) struct CachedMetaReaderFactory { store: Arc, + store_url: ObjectStoreUrl, } impl CachedMetaReaderFactory { - pub(crate) fn new(store: Arc) -> Self { - Self { store } + pub(crate) fn new(store: Arc, store_url: ObjectStoreUrl) -> Self { + Self { store, store_url } + } + + pub(crate) fn object_store_url(&self) -> &ObjectStoreUrl { + &self.store_url } pub(crate) fn create_liquid_reader( @@ -62,18 +64,13 @@ impl CachedMetaReaderFactory { metrics: &ExecutionPlanMetricsSet, ) -> ParquetMetadataCacheReader { let path = partitioned_file.object_meta.location.clone(); - let store = Arc::clone(&self.store); - #[allow(deprecated)] - let mut inner = ParquetObjectReader::new(store, path.clone()) - .with_file_size(partitioned_file.object_meta.size); - - if let Some(hint) = metadata_size_hint { - inner = inner.with_footer_size_hint(hint); - } ParquetMetadataCacheReader { file_metrics: ParquetFileMetrics::new(partition_index, path.as_ref(), metrics), - inner, + store: Arc::clone(&self.store), + store_url: self.store_url.clone(), + file_size: partitioned_file.object_meta.size, + metadata_size_hint, path, } } @@ -98,7 +95,7 @@ impl ParquetFileReaderFactory for CachedMetaReaderFactory { } struct MetadataCache { - val: RwLock>>, + val: RwLock>>, } impl MetadataCache { @@ -112,11 +109,17 @@ impl MetadataCache { #[derive(Clone)] pub struct ParquetMetadataCacheReader { file_metrics: ParquetFileMetrics, - #[allow(deprecated)] - inner: ParquetObjectReader, + store: Arc, + store_url: ObjectStoreUrl, + file_size: u64, + metadata_size_hint: Option, path: Path, } +fn to_parquet_err(error: object_store::Error) -> ParquetError { + ParquetError::External(Box::new(error)) +} + impl AsyncFileReader for ParquetMetadataCacheReader { fn get_byte_ranges( &mut self, @@ -124,41 +127,57 @@ impl AsyncFileReader for ParquetMetadataCacheReader { ) -> BoxFuture<'_, parquet::errors::Result>> { let total: u64 = ranges.iter().map(|r| r.end - r.start).sum(); self.file_metrics.bytes_scanned.add(total as usize); - self.inner.get_byte_ranges(ranges) + async move { + self.store + .get_ranges(&self.path, &ranges) + .await + .map_err(to_parquet_err) + } + .boxed() } fn get_bytes(&mut self, range: Range) -> BoxFuture<'_, parquet::errors::Result> { self.file_metrics .bytes_scanned .add((range.end - range.start) as usize); - self.inner.get_bytes(range) + async move { + self.store + .get_range(&self.path, range) + .await + .map_err(to_parquet_err) + } + .boxed() } fn get_metadata( &mut self, options: Option<&ArrowReaderOptions>, ) -> BoxFuture<'_, parquet::errors::Result>> { - let path = self.path.clone(); + let cache_key = (self.store_url.clone(), self.path.clone()); let options = options.cloned(); async move { // First check with read lock { let cache = META_CACHE.val.read().await; - if let Some(meta) = cache.get(&path) { + if let Some(meta) = cache.get(&cache_key) { return Ok(meta.clone()); } } // Upgrade to write lock and double-check let mut cache = META_CACHE.val.write().await; - match cache.entry(path.clone()) { + match cache.entry(cache_key) { std::collections::hash_map::Entry::Occupied(entry) => Ok(entry.get().clone()), std::collections::hash_map::Entry::Vacant(entry) => { - let meta = self.inner.get_metadata(options.as_ref()).await?; - let meta = Arc::try_unwrap(meta).unwrap_or_else(|e| e.as_ref().clone()); + let file_size = self.file_size; + let meta = ParquetMetaDataReader::new() + .with_arrow_reader_options(options.as_ref()) + .with_prefetch_hint(self.metadata_size_hint) + .load_and_finish(&mut *self, file_size) + .await?; let mut reader = ParquetMetaDataReader::new_with_metadata(meta.clone()) .with_page_index_policy(PageIndexPolicy::Optional); - reader.load_page_index(&mut self.inner).await?; + reader.load_page_index(&mut *self).await?; let meta = Arc::new(reader.finish()?); entry.insert(meta.clone()); Ok(meta) @@ -174,14 +193,13 @@ impl AsyncFileReader for ParquetMetadataCacheReader { pub struct LiquidParquetSource { metrics: ExecutionPlanMetricsSet, predicate: Option>, - pruning_predicate: Option>, - page_pruning_predicate: Option>, table_parquet_options: TableParquetOptions, liquid_cache: LiquidCacheParquetRef, projection: ProjectionExprs, table_schema: TableSchema, span: Option>, squeeze_hints: Arc, + prefetch: bool, } impl LiquidParquetSource { @@ -214,45 +232,20 @@ impl LiquidParquetSource { } } + /// Enable or disable row-group prefetching. + pub fn with_prefetch(mut self, prefetch: bool) -> Self { + self.prefetch = prefetch; + self + } + /// The typed squeeze hints currently attached to this source. pub fn squeeze_hints(&self) -> &Arc { &self.squeeze_hints } - /// Set predicate information, also sets pruning_predicate and page_pruning_predicate attributes - pub fn with_predicate( - mut self, - file_schema: Arc, - predicate: Arc, - ) -> Self { - let metrics = ExecutionPlanMetricsSet::new(); - let predicate_creation_errors = - MetricBuilder::new(&metrics).global_counter("num_predicate_creation_errors"); - - self.metrics = metrics; - self.predicate = Some(Arc::clone(&predicate)); - - match PruningPredicateBuilder::new() - .with_file_schema(Arc::clone(&file_schema)) - .try_build(Arc::clone(&predicate)) - { - Ok(pruning_predicate) => { - if !pruning_predicate.always_true() { - self.pruning_predicate = Some(Arc::new(pruning_predicate)); - } - } - Err(e) => { - log::debug!("Could not create pruning predicate for: {e}"); - predicate_creation_errors.add(1); - } - }; - - let page_pruning_predicate = Arc::new(PagePruningAccessPlanFilter::new( - &predicate, - Arc::clone(&file_schema), - )); - self.page_pruning_predicate = Some(page_pruning_predicate); - + /// Set predicate information. + pub fn with_predicate(mut self, predicate: Arc) -> Self { + self.predicate = Some(predicate); self } @@ -261,7 +254,6 @@ impl LiquidParquetSource { let predicate = source.filter(); let table_schema = source.table_schema().clone(); - let file_schema = table_schema.file_schema().clone(); let projection = source.projection().cloned().unwrap_or_else(|| { let table_schema = table_schema.table_schema(); ProjectionExprs::from_indices( @@ -276,14 +268,13 @@ impl LiquidParquetSource { projection, metrics: source.metrics().clone(), predicate: None, - pruning_predicate: None, - page_pruning_predicate: None, span: None, squeeze_hints: Arc::default(), + prefetch: true, }; if let Some(predicate) = predicate { - v = v.with_predicate(file_schema, predicate); + v = v.with_predicate(predicate); } v @@ -297,38 +288,56 @@ impl LiquidParquetSource { impl FileSource for LiquidParquetSource { fn create_file_opener( + &self, + _object_store: Arc, + _base_config: &FileScanConfig, + _partition: usize, + ) -> Result> { + internal_err!( + "LiquidParquetSource::create_file_opener called but it supports the Morsel API, please use that instead" + ) + } + + fn create_morselizer( &self, object_store: Arc, base_config: &FileScanConfig, partition: usize, - ) -> Result> { + ) -> Result> { let expr_adapter_factory = base_config .expr_adapter_factory .clone() .unwrap_or_else(|| Arc::new(DefaultPhysicalExprAdapterFactory) as _); - let reader_factory = Arc::new(CachedMetaReaderFactory::new(object_store)); + let reader_factory = Arc::new(CachedMetaReaderFactory::new( + object_store, + base_config.object_store_url.clone(), + )); let execution_span = self .span .clone() .map(|span| fastrace::Span::enter_with_parent(format!("opener_{partition}"), &span)); - let opener = LiquidParquetOpener::new( - partition, - self.projection.clone(), - base_config.limit, - self.predicate.clone(), - self.table_schema.clone(), - self.metrics.clone(), - self.liquid_cache.clone(), - reader_factory, - self.reorder_filters(), + Ok(Box::new(LiquidMorselizer { + partition_index: partition, + projection: self.projection.clone(), + // From the cache, not the session config: the reader indexes the + // cache by batch id and the parquet fallback turns that id back into + // rows with the cache batch size, so the two must be the same number + // (issue #13). Sourcing it here makes the reader's own + // `debug_assert_eq!` hold by construction instead of by coincidence. + batch_size: self.liquid_cache.batch_size(), + predicate: self.predicate.clone(), + table_schema: self.table_schema.clone(), + metrics: self.metrics.clone(), + liquid_cache: self.liquid_cache.clone(), + parquet_file_reader_factory: reader_factory, + reorder_filters: self.reorder_filters(), expr_adapter_factory, - execution_span.map(Arc::new), - Arc::clone(&self.squeeze_hints), - ); - - Ok(Arc::new(opener)) + span: execution_span.map(Arc::new), + squeeze_hints: Arc::clone(&self.squeeze_hints), + prefetch: self.prefetch, + })) } /// Deliberately ignores the requested batch size: the reader indexes the cache @@ -349,6 +358,10 @@ impl FileSource for LiquidParquetSource { Arc::new(self.clone()) } + fn filter(&self) -> Option> { + self.predicate.clone() + } + fn table_schema(&self) -> &TableSchema { &self.table_schema } @@ -374,6 +387,58 @@ impl FileSource for LiquidParquetSource { "liquid_parquet" } + fn fmt_extra(&self, t: DisplayFormatType, f: &mut Formatter) -> fmt::Result { + match t { + DisplayFormatType::Default | DisplayFormatType::Verbose => { + if let Some(predicate) = self.filter() { + write!(f, ", predicate={predicate}")?; + } + Ok(()) + } + DisplayFormatType::TreeRender => Ok(()), + } + } + + fn try_pushdown_filters( + &self, + filters: Vec>, + _config: &ConfigOptions, + ) -> Result>> { + let filters: Vec<_> = filters + .into_iter() + .map(|filter| { + if can_expr_be_pushed_down_with_schemas(&filter, self.table_schema.file_schema()) { + PushedDownPredicate::supported(filter) + } else { + PushedDownPredicate::unsupported(filter) + } + }) + .collect(); + + if filters + .iter() + .all(|filter| matches!(filter.discriminant, PushedDown::No)) + { + return Ok(FilterPushdownPropagation::with_parent_pushdown_result( + vec![PushedDown::No; filters.len()], + )); + } + + let supported = filters + .iter() + .filter_map(|filter| match filter.discriminant { + PushedDown::Yes => Some(Arc::clone(&filter.predicate)), + PushedDown::No => None, + }); + let predicate = conjunction(self.predicate.iter().cloned().chain(supported)); + let source = Arc::new(self.clone().with_predicate(predicate)); + + Ok(FilterPushdownPropagation::with_parent_pushdown_result( + filters.iter().map(|filter| filter.discriminant).collect(), + ) + .with_updated_node(source)) + } + fn apply_expressions( &self, f: &mut dyn FnMut(&Arc) -> Result, @@ -381,7 +446,7 @@ impl FileSource for LiquidParquetSource { apply_expression_roots( self.predicate .iter() - .chain(self.projection.iter().map(|proj_expr| &proj_expr.expr)), + .chain(self.projection.iter().map(|projection| &projection.expr)), f, ) } diff --git a/src/datafusion/src/reader/runtime/liquid_cache_reader.rs b/src/datafusion/src/reader/runtime/liquid_cache_reader.rs index fadc8e70d..c0ae19c9c 100644 --- a/src/datafusion/src/reader/runtime/liquid_cache_reader.rs +++ b/src/datafusion/src/reader/runtime/liquid_cache_reader.rs @@ -7,10 +7,10 @@ use arrow::array::{Array, ArrayRef, BooleanArray, RecordBatch}; use arrow::buffer::BooleanBuffer; use arrow::compute::prep_null_mask_filter; use arrow::record_batch::RecordBatchOptions; -use arrow_schema::{ArrowError, Schema, SchemaRef}; +use arrow_schema::{ArrowError, SchemaRef}; use futures::{Stream, StreamExt, future::BoxFuture, stream::BoxStream}; use parquet::arrow::arrow_reader::{ - ArrowPredicate, ArrowReaderMetadata, ArrowReaderOptions, RowSelection, RowSelector, + ArrowReaderMetadata, ArrowReaderOptions, RowSelection, RowSelector, }; use parquet::arrow::{ParquetRecordBatchStreamBuilder, ProjectionMask}; use parquet::errors::ParquetError; @@ -78,7 +78,7 @@ pub(crate) struct ParquetFallbackConfig { pub(crate) row_count: usize, } -struct ParquetFallback { +pub(crate) struct ParquetFallback { row_group_idx: usize, metadata: Arc, input: ParquetMetadataCacheReader, @@ -92,6 +92,11 @@ struct ParquetFallback { impl LiquidCacheReader { pub(crate) fn new(config: LiquidCacheReaderConfig) -> Self { + debug_assert_eq!( + config.batch_size, + config.cached_row_group.batch_size(), + "DataFusion and LiquidCache batch sizes must agree" + ); let inner = LiquidCacheReaderInner::new( config.batch_size, config.selection, @@ -105,14 +110,6 @@ impl LiquidCacheReader { row_filter: config.row_filter, } } - - pub(crate) fn into_filter(self) -> Option { - debug_assert!( - matches!(self.state, ReaderState::Finished), - "cannot extract filter before reader completes" - ); - self.row_filter - } } impl Stream for LiquidCacheReader { @@ -161,7 +158,7 @@ impl Stream for LiquidCacheReader { } impl ParquetFallback { - fn new(config: ParquetFallbackConfig) -> Self { + pub(crate) fn new(config: ParquetFallbackConfig) -> Self { Self { row_group_idx: config.row_group_idx, metadata: config.metadata, @@ -175,7 +172,10 @@ impl ParquetFallback { } } - async fn fetch_batch(&mut self, batch_id: BatchID) -> Result { + pub(crate) async fn fetch_batch( + &mut self, + batch_id: BatchID, + ) -> Result { if self.stream.is_none() || batch_id != self.next_batch_id { self.rebuild_stream(batch_id)?; } @@ -299,43 +299,62 @@ impl LiquidCacheReaderInner { row_filter: &mut Option, selection: Vec, ) -> Result { - let mut input_selection = row_selector_to_boolean_buffer(&selection); + let input_selection = row_selector_to_boolean_buffer(&selection); + + if let Some(snapshot_selection) = self + .cached_row_group + .snapshot_selection(self.current_batch_id) + { + // A plain AND, not `boolean_buffer_and_then`: the prefetch stored this + // snapshot against the *same* window this reader is looking at (both + // come from `take_next_batch` over the one `RowSelection`, at the cache + // batch size), so both buffers index rows of the window absolutely. + // `and_then` instead reads its right operand as positions among the + // left's set bits, and asserts `left.count_set_bits() == right.len()`. + // That holds only while every row of the window is selected; page index + // pruning clears most bits without shortening the window, and the + // assertion then trips on any pruned, filtered scan. + // + // Release builds were not returning wrong rows: equal lengths took + // `boolean_buffer_and_then`'s early return, which yields the right + // operand, and the snapshot is `apply_predicates(input_selection)` and + // so already a subset. The AND says that directly instead of relying on + // a shortcut inside a function whose contract this call does not meet. + debug_assert_eq!( + input_selection.len(), + snapshot_selection.len(), + "the snapshot and the reader must window the row group identically" + ); + return Ok(&input_selection & &snapshot_selection); + } let Some(filter) = row_filter.as_mut() else { return Ok(input_selection); }; - for predicate in filter.predicates_mut() { - if input_selection.count_set_bits() == 0 { - break; - } - - let boolean_array = match self - .cached_row_group - .evaluate_selection_with_predicate( - self.current_batch_id, - &input_selection, - predicate, - ) - .await - { - Some(result) => result?, - None => { - self.evaluate_predicate_after_materialize(&input_selection, predicate) - .await? - } - }; - - let boolean_mask = if boolean_array.null_count() == 0 { - boolean_array.into_parts().0 - } else { - prep_null_mask_filter(&boolean_array).into_parts().0 - }; - - input_selection = boolean_buffer_and_then(&input_selection, &boolean_mask); + if let Some(selection) = apply_predicates( + &self.cached_row_group, + self.current_batch_id, + input_selection.clone(), + filter, + ) + .await? + { + return Ok(selection); } - Ok(input_selection) + self.read_parquet_batch_and_fill_cache(self.current_batch_id) + .await?; + apply_predicates( + &self.cached_row_group, + self.current_batch_id, + input_selection, + filter, + ) + .await? + .ok_or_else(|| { + ArrowError::ComputeError("predicate unavailable after materialization".to_string()) + }) } #[fastrace::trace] @@ -423,9 +442,11 @@ impl LiquidCacheReaderInner { })?; let array = Arc::clone(record_batch.column(col_idx)); - match column.insert(batch_id, array).await { + match column.insert(batch_id, Arc::clone(&array)).await { Ok(()) | Err(InsertArrowArrayError::AlreadyCached) => {} - Err(InsertArrowArrayError::CacheFull) => {} + Err(InsertArrowArrayError::CacheFull) => { + column.insert_snapshot(batch_id, array); + } } } @@ -433,57 +454,6 @@ impl LiquidCacheReaderInner { Ok(record_batch) } - async fn evaluate_predicate_after_materialize( - &mut self, - selection: &BooleanBuffer, - predicate: &mut crate::reader::LiquidPredicate, - ) -> Result { - let record_batch = self - .read_parquet_batch_and_fill_cache(self.current_batch_id) - .await?; - - if let Some(result) = self - .cached_row_group - .evaluate_selection_with_predicate(self.current_batch_id, selection, predicate) - .await - { - return result; - } - - let column_ids = predicate.predicate_column_ids(); - let mut arrays = Vec::with_capacity(column_ids.len()); - let mut fields = Vec::with_capacity(column_ids.len()); - - for column_id in column_ids { - let array = self.parquet_array(&record_batch, column_id)?; - arrays.push(filter_array(array, selection)?); - - let field = self - .cached_row_group - .get_column(column_id as u64) - .ok_or_else(|| { - ArrowError::ComputeError(format!( - "column {column_id} not present in liquid cache" - )) - })? - .field() - .as_ref() - .clone(); - fields.push(field); - } - - let schema = Arc::new(Schema::new(fields)); - let predicate_batch = if arrays.is_empty() { - let options = - RecordBatchOptions::new().with_row_count(Some(selection.count_set_bits())); - RecordBatch::try_new_with_options(schema, arrays, &options)? - } else { - RecordBatch::try_new(schema, arrays)? - }; - - predicate.evaluate(predicate_batch) - } - fn parquet_array( &self, record_batch: &RecordBatch, @@ -504,6 +474,35 @@ impl LiquidCacheReaderInner { } } +pub(crate) async fn apply_predicates( + row_group: &CachedRowGroupRef, + batch_id: BatchID, + mut input_selection: BooleanBuffer, + filter: &mut LiquidRowFilter, +) -> Result, ArrowError> { + for predicate in filter.predicates_mut() { + if input_selection.count_set_bits() == 0 { + break; + } + + let Some(boolean_array) = row_group + .evaluate_selection_with_predicate(batch_id, &input_selection, predicate) + .await + else { + return Ok(None); + }; + let boolean_array = boolean_array?; + let boolean_mask = if boolean_array.null_count() == 0 { + boolean_array.into_parts().0 + } else { + prep_null_mask_filter(&boolean_array).into_parts().0 + }; + input_selection = boolean_buffer_and_then(&input_selection, &boolean_mask); + } + + Ok(Some(input_selection)) +} + fn filter_array(array: ArrayRef, selection: &BooleanBuffer) -> Result { let selection_array = BooleanArray::new(selection.clone(), None); arrow::compute::filter(array.as_ref(), &selection_array) @@ -591,12 +590,11 @@ mod tests { std::fs::metadata(&parquet_path).unwrap().len(), ); let metrics = ExecutionPlanMetricsSet::new(); - let input = CachedMetaReaderFactory::new(object_store).create_liquid_reader( - 0, - partitioned_file, - None, - &metrics, - ); + let input = CachedMetaReaderFactory::new( + object_store, + datafusion::execution::object_store::ObjectStoreUrl::parse("test-runtime:///").unwrap(), + ) + .create_liquid_reader(0, partitioned_file, None, &metrics); let projection = ProjectionMask::roots( reader_metadata.metadata().file_metadata().schema_descr(), [0], @@ -613,7 +611,14 @@ mod tests { Box::new(AlwaysHydrate::new()), ) .await; - let file = cache.register_or_get_file("test".to_string(), schema.clone()); + let file = cache.register_or_get_file( + crate::cache::ParquetFileIdentity::new( + datafusion::execution::object_store::ObjectStoreUrl::parse("test-runtime:///") + .unwrap(), + "test".to_string(), + ), + schema.clone(), + ); let row_group = file.create_row_group(0, vec![]); let column = row_group.get_column(0).unwrap(); @@ -753,30 +758,6 @@ mod tests { assert_eq!(batch.num_rows(), 2); } - #[tokio::test] - async fn into_filter_returns_stored_filter_after_completion() { - let batch_size = 2; - let test = make_row_group(batch_size, &[vec![1, 2]]).await; - let selection = RowSelection::from(Vec::::new()); - let filter = LiquidRowFilter::new(Vec::new()); - - let mut reader = test.reader(ReaderRequest { - selection, - row_filter: Some(filter), - projection_columns: vec![0], - schema: Arc::clone(&test.schema), - }); - - let waker = futures::task::noop_waker(); - let mut cx = Context::from_waker(&waker); - assert!(matches!( - Pin::new(&mut reader).poll_next(&mut cx), - Poll::Ready(None) - )); - - assert!(reader.into_filter().is_some()); - } - #[tokio::test] async fn predicate_filters_rows_across_batches() { let batches = vec![vec![1, 2], vec![3, 4]]; diff --git a/src/datafusion/src/reader/runtime/liquid_predicate.rs b/src/datafusion/src/reader/runtime/liquid_predicate.rs index 2cae21e29..405f59c07 100644 --- a/src/datafusion/src/reader/runtime/liquid_predicate.rs +++ b/src/datafusion/src/reader/runtime/liquid_predicate.rs @@ -46,21 +46,25 @@ fn extract_column_literal(expr: &Arc) -> Option<(&str, Arc() && binary.right().is::() { - return extract_column_literal(binary.left()); + return column_name(binary.left()).map(|name| (name, Arc::clone(expr))); } else if let Some(like_expr) = expr.downcast_ref::() && like_expr.pattern().is::() { - return extract_column_literal(like_expr.expr()); - } else if let Some(cast_expr) = expr.downcast_ref::() { - return extract_column_literal(cast_expr.expr()); - } else if let Some(try_cast_expr) = expr.downcast_ref::() { - return extract_column_literal(try_cast_expr.expr()); - } else if let Some(column) = expr.downcast_ref::() { - return Some((column.name(), Arc::clone(expr))); + return column_name(like_expr.expr()).map(|name| (name, Arc::clone(expr))); } None } +fn column_name(expr: &Arc) -> Option<&str> { + if let Some(cast_expr) = expr.downcast_ref::() { + column_name(cast_expr.expr()) + } else if let Some(try_cast_expr) = expr.downcast_ref::() { + column_name(try_cast_expr.expr()) + } else { + expr.downcast_ref::().map(Column::name) + } +} + #[cfg(test)] mod tests { use super::*; diff --git a/src/datafusion/src/reader/runtime/liquid_stream.rs b/src/datafusion/src/reader/runtime/liquid_stream.rs deleted file mode 100644 index 70da75766..000000000 --- a/src/datafusion/src/reader/runtime/liquid_stream.rs +++ /dev/null @@ -1,1047 +0,0 @@ -use crate::cache::{CachedFileRef, CachedRowGroupRef}; -use crate::reader::plantime::{LiquidRowFilter, ParquetMetadataCacheReader}; -use arrow::array::RecordBatch; -use arrow_schema::{Schema, SchemaRef}; -use fastrace::Event; -use fastrace::local::LocalSpan; -use futures::Stream; -use parquet::{ - arrow::{ - ProjectionMask, - arrow_reader::{ArrowPredicate, RowSelection, RowSelector}, - }, - errors::ParquetError, - file::metadata::ParquetMetaData, -}; -use std::{ - collections::VecDeque, - fmt::Formatter, - pin::Pin, - sync::Arc, - task::{Context, Poll}, -}; - -use super::liquid_cache_reader::{ - LiquidCacheReader, LiquidCacheReaderConfig, ParquetFallbackConfig, -}; -use super::utils::{get_root_column_ids, limit_row_selection, offset_row_selection}; - -type PlanResult = Option; - -struct ReaderFactory { - metadata: Arc, - - input: ParquetMetadataCacheReader, - - filter: Option, - - limit: Option, - - offset: Option, - - cached_file: CachedFileRef, -} - -impl ReaderFactory { - /// Plans what to read from cache vs parquet for the next row group - fn plan_row_group( - &mut self, - row_group_idx: usize, - selection: Option, - projection: ProjectionMask, - ) -> PlanResult { - let meta = self.metadata.row_group(row_group_idx); - - let mut predicate_projection: Option = None; - if let Some(filter) = self.filter.as_mut() { - for predicate in filter.predicates_mut() { - let p_projection = predicate.projection(); - if let Some(ref mut p) = predicate_projection { - p.union(p_projection); - } else { - predicate_projection = Some(p_projection.clone()); - } - } - } - - let mut selection = - selection.unwrap_or_else(|| vec![RowSelector::select(meta.num_rows() as usize)].into()); - - let rows_before = selection.row_count(); - - if rows_before == 0 { - return None; - } - - if let Some(offset) = self.offset { - selection = offset_row_selection(selection, offset); - } - - if let Some(limit) = self.limit { - selection = limit_row_selection(selection, limit); - } - - let rows_after = selection.row_count(); - - // Update offset if necessary - if let Some(offset) = &mut self.offset { - // Reduction is either because of offset or limit, as limit is applied - // after offset has been "exhausted" can just use saturating sub here - *offset = offset.saturating_sub(rows_before - rows_after) - } - - if rows_after == 0 { - return None; - } - - if let Some(limit) = &mut self.limit { - *limit -= rows_after; - } - - let mut cache_projection = projection.clone(); - if let Some(ref predicate_projection) = predicate_projection { - cache_projection.union(predicate_projection); - } - - let schema_descr = self.metadata.file_metadata().schema_descr(); - let cache_column_ids = get_root_column_ids(schema_descr, &cache_projection); - let predicate_column_ids = if let Some(ref predicate_projection) = predicate_projection { - get_root_column_ids(schema_descr, predicate_projection) - } else { - Vec::new() - }; - let cached_row_group = self - .cached_file - .create_row_group(row_group_idx as u64, predicate_column_ids); - - let projection_column_ids = get_root_column_ids(schema_descr, &projection); - - let context = PlanningContext { - row_group_idx, - selection, - cached_row_group, - cache_projection, - projection_column_ids, - cache_column_ids, - }; - - Some(context) - } -} - -fn build_projection_schema(file_schema: &SchemaRef, projection_column_ids: &[usize]) -> SchemaRef { - let fields: Vec<_> = projection_column_ids - .iter() - .filter_map(|column_id| file_schema.fields().get(*column_id)) - .map(|field_ref| field_ref.as_ref().clone()) - .collect(); - Arc::new(Schema::new(fields)) -} - -/// Context for planning what to read from cache vs parquet -struct PlanningContext { - row_group_idx: usize, - selection: RowSelection, - cached_row_group: CachedRowGroupRef, - cache_projection: ProjectionMask, - projection_column_ids: Vec, - cache_column_ids: Vec, -} - -fn build_liquid_cache_reader( - reader_factory: &mut ReaderFactory, - context: PlanningContext, - schema: SchemaRef, -) -> LiquidCacheReader { - let row_count = reader_factory - .metadata - .row_group(context.row_group_idx) - .num_rows() as usize; - // `current_batch_id` counts read batches, and the parquet fallback multiplies - // it by `cache_batch_size` to get a row offset, so the read size and the cache - // size must be the same number. Read it once and use it for both rather than - // deriving them separately — issue #13 was the two drifting apart. - let cache_batch_size = context.cached_row_group.batch_size(); - LiquidCacheReader::new(LiquidCacheReaderConfig { - batch_size: cache_batch_size, - selection: context.selection, - row_filter: reader_factory.filter.take(), - cached_row_group: context.cached_row_group, - projection_columns: context.projection_column_ids, - schema, - parquet_fallback: ParquetFallbackConfig { - row_group_idx: context.row_group_idx, - metadata: Arc::clone(&reader_factory.metadata), - input: reader_factory.input.clone(), - cache_projection: context.cache_projection, - cache_column_ids: context.cache_column_ids, - cache_batch_size, - row_count, - }, - }) -} - -enum StreamState { - /// At the start of a new row group, or the end of the parquet stream - Init, - /// Decoding a batch from cache - ReadFromCache(Box), - /// A decode error ended the scan. Terminal: the stream yields `None` from - /// here on and never plans another row group. - Finished, -} - -impl std::fmt::Debug for StreamState { - fn fmt(&self, f: &mut Formatter<'_>) -> std::fmt::Result { - match self { - StreamState::Init => write!(f, "StreamState::Init"), - StreamState::ReadFromCache(_) => write!(f, "StreamState::Decoding"), - StreamState::Finished => write!(f, "StreamState::Finished"), - } - } -} - -pub struct LiquidStreamBuilder { - pub(crate) input: ParquetMetadataCacheReader, - - pub(crate) metadata: Arc, - - pub(crate) row_groups: Option>, - - pub(crate) projection: ProjectionMask, - - pub(crate) filter: Option, - - pub(crate) selection: Option, - - pub(crate) limit: Option, - - pub(crate) offset: Option, - - pub(crate) span: Option, -} - -impl LiquidStreamBuilder { - pub fn new(input: ParquetMetadataCacheReader, metadata: Arc) -> Self { - Self { - input, - metadata, - row_groups: None, - projection: ProjectionMask::all(), - filter: None, - selection: None, - limit: None, - offset: None, - span: None, - } - } - - pub fn with_row_groups(mut self, row_groups: Vec) -> Self { - self.row_groups = Some(row_groups); - self - } - - pub fn with_projection(mut self, projection: ProjectionMask) -> Self { - self.projection = projection; - self - } - - pub fn with_selection(mut self, selection: Option) -> Self { - self.selection = selection; - self - } - - pub fn with_limit(mut self, limit: Option) -> Self { - self.limit = limit; - self - } - - pub fn with_row_filter(mut self, filter: LiquidRowFilter) -> Self { - self.filter = Some(filter); - self - } - - pub fn with_span(mut self, span: fastrace::Span) -> Self { - self.span = Some(span); - self - } - - pub fn build(self, liquid_cache: CachedFileRef) -> Result { - let num_row_groups = self.metadata.row_groups().len(); - - let row_groups: VecDeque = match self.row_groups { - Some(row_groups) => { - if let Some(col) = row_groups.iter().find(|x| **x >= num_row_groups) { - return Err(ParquetError::ArrowError(format!( - "row group {col} out of bounds 0..{num_row_groups}" - ))); - } - row_groups.into() - } - None => (0..self.metadata.row_groups().len()).collect(), - }; - - // No batch size is chosen here. The scan must read in cache-sized batches - // (see `build_liquid_cache_reader`), so the size is read from the cached row - // group at the one place it is used, and the session config never reaches - // the reader at all. - let schema_descr = self.metadata.file_metadata().schema_descr(); - let projection_column_ids = get_root_column_ids(schema_descr, &self.projection); - let file_schema = liquid_cache.schema(); - let schema = build_projection_schema(&file_schema, &projection_column_ids); - - // `plan_row_group` applies limit/offset by truncating the row - // selection BEFORE the row filter runs, so combining them with a - // filter caps the rows *scanned* rather than the rows *matched* — - // silently dropping matches that sit past the first `limit + offset` - // physical rows (upstream parquet counts the limit against - // post-filter matches instead). Until limit accounting moves after - // predicate evaluation, only honor limit/offset for unfiltered - // scans, where scanned rows == emitted rows and truncation is - // exact. Filtered scans still get capped post-filter by - // DataFusion's FileStream, which slices emitted batches against - // `FileScanConfig::limit`; all that is lost is scan-internal early - // termination. - let (limit, offset) = if self.filter.is_some() { - (None, None) - } else { - (self.limit, self.offset) - }; - - let reader = ReaderFactory { - metadata: Arc::clone(&self.metadata), - input: self.input, - filter: self.filter, - limit, - offset, - cached_file: liquid_cache, - }; - - Ok(LiquidStream { - metadata: self.metadata, - schema, - row_groups, - projection: self.projection, - selection: self.selection, - reader: Some(reader), - state: StreamState::Init, - span: self.span, - }) - } -} - -pub struct LiquidStream { - metadata: Arc, - - schema: SchemaRef, - - row_groups: VecDeque, - - projection: ProjectionMask, - - selection: Option, - - /// This is an option so it can be moved into a future - reader: Option, - - state: StreamState, - - span: Option, -} - -impl std::fmt::Debug for LiquidStream { - fn fmt(&self, f: &mut Formatter<'_>) -> std::fmt::Result { - f.debug_struct("ParquetRecordBatchStream") - .field("metadata", &self.metadata) - .field("schema", &self.schema) - .field("projection", &self.projection) - .field("state", &self.state) - .finish() - } -} - -impl LiquidStream { - pub fn schema(&self) -> &SchemaRef { - &self.schema - } -} - -impl Stream for LiquidStream { - type Item = Result; - - fn poll_next(mut self: Pin<&mut Self>, cx: &mut Context<'_>) -> Poll> { - let _guard = self.span.as_ref().map(|s| s.set_local_parent()); - loop { - let state = std::mem::replace(&mut self.state, StreamState::Init); - - match state { - StreamState::ReadFromCache(mut batch_reader) => { - match Pin::new(&mut *batch_reader).poll_next(cx) { - Poll::Ready(Some(Ok(batch))) => { - self.state = StreamState::ReadFromCache(batch_reader); - return Poll::Ready(Some(Ok(batch))); - } - Poll::Ready(Some(Err(e))) => { - // A decode fault belongs to this one query. Panicking - // here crosses the host's task boundary and takes down - // a worker thread, so return the error instead. - // - // Terminal, and that matters for correctness: the reader - // emits this error without incrementing `current_batch_id` - // and leaves itself `Ready`, so resuming would read the - // next window against a stale batch id. - self.state = StreamState::Finished; - return Poll::Ready(Some(Err(e.into()))); - } - Poll::Ready(None) => { - let batch_reader = *batch_reader; - let filter = batch_reader.into_filter(); - self.reader.as_mut().unwrap().filter = filter; - // state left as Init, continue loop to plan next row group - } - Poll::Pending => { - self.state = StreamState::ReadFromCache(batch_reader); - return Poll::Pending; - } - } - } - StreamState::Finished => { - self.state = StreamState::Finished; - return Poll::Ready(None); - } - StreamState::Init => { - let row_group_idx = match self.row_groups.pop_front() { - Some(idx) => idx, - None => return Poll::Ready(None), - }; - - let row_count = self.metadata.row_group(row_group_idx).num_rows() as usize; - - let selection = self.selection.as_mut().map(|s| s.split_off(row_count)); - - LocalSpan::add_event(Event::new("LiquidStream::plan_row_group")); - let projection = self.projection.clone(); - let maybe_context = self.reader.as_mut().expect("lost reader").plan_row_group( - row_group_idx, - selection, - projection, - ); - match maybe_context { - Some(context) => { - LocalSpan::add_event(Event::new("LiquidStream::read_from_cache")); - let schema = Arc::clone(&self.schema); - let reader_factory = self.reader.as_mut().unwrap(); - let batch_reader = - build_liquid_cache_reader(reader_factory, context, schema); - self.state = StreamState::ReadFromCache(Box::new(batch_reader)); - } - None => { - self.state = StreamState::Init; - } - } - } - } - } - } -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::cache::{BatchID, CachedFileRef, LiquidCacheParquet}; - use crate::reader::plantime::{ - CachedMetaReaderFactory, FilterCandidateBuilder, LiquidPredicate, - }; - use arrow::array::{Array, ArrayRef, Int32Array}; - use arrow_schema::{DataType, Field, Schema}; - use datafusion::common::ScalarValue; - use datafusion::datasource::listing::PartitionedFile; - use datafusion::logical_expr::Operator; - use datafusion::physical_expr::PhysicalExpr; - use datafusion::physical_expr::expressions::{BinaryExpr, Column, Literal}; - use datafusion::physical_plan::metrics::ExecutionPlanMetricsSet; - use futures::StreamExt; - use liquid_cache::cache::AlwaysHydrate; - use liquid_cache::cache::squeeze_policies::Evict; - use liquid_cache::cache_policies::LiquidPolicy; - use object_store::local::LocalFileSystem; - use parquet::arrow::ArrowWriter; - use parquet::arrow::arrow_reader::{ArrowReaderMetadata, ArrowReaderOptions}; - use std::fs::File; - use std::sync::Arc; - - use crate::test_utils::mount_test_store as test_mount; - - fn write_two_row_group_file(path: &std::path::Path, schema: SchemaRef) { - let file = File::create(path).unwrap(); - let mut writer = ArrowWriter::try_new(file, schema.clone(), None).unwrap(); - let batch0 = RecordBatch::try_new( - schema.clone(), - vec![ - Arc::new(Int32Array::from(vec![0, 1, 2, 3])), - Arc::new(Int32Array::from(vec![10, 11, 12, 13])), - ], - ) - .unwrap(); - let batch1 = RecordBatch::try_new( - schema, - vec![ - Arc::new(Int32Array::from(vec![4, 5, 6, 7])), - Arc::new(Int32Array::from(vec![14, 15, 16, 17])), - ], - ) - .unwrap(); - writer.write(&batch0).unwrap(); - writer.flush().unwrap(); - writer.write(&batch1).unwrap(); - writer.close().unwrap(); - } - - fn write_single_row_group_file(path: &std::path::Path, schema: SchemaRef, a: Vec) { - let file = File::create(path).unwrap(); - let mut writer = ArrowWriter::try_new(file, schema.clone(), None).unwrap(); - let b: Vec<_> = a.iter().map(|value| value + 1000).collect(); - let batch = RecordBatch::try_new( - schema, - vec![Arc::new(Int32Array::from(a)), Arc::new(Int32Array::from(b))], - ) - .unwrap(); - writer.write(&batch).unwrap(); - writer.close().unwrap(); - } - - async fn make_liquid_stream( - max_memory_bytes: usize, - max_disk_bytes: usize, - row_filter: Option, - ) -> ( - LiquidStream, - Arc, - CachedFileRef, - tempfile::TempDir, - ) { - let schema = Arc::new(Schema::new(vec![ - Field::new("a", DataType::Int32, false), - Field::new("b", DataType::Int32, false), - ])); - let tmp_dir = tempfile::tempdir().unwrap(); - let parquet_path = tmp_dir.path().join("data.parquet"); - write_two_row_group_file(&parquet_path, schema.clone()); - let metadata_file = File::open(&parquet_path).unwrap(); - let reader_metadata = - ArrowReaderMetadata::load(&metadata_file, ArrowReaderOptions::new()).unwrap(); - let object_store = Arc::new(LocalFileSystem::new_with_prefix(tmp_dir.path()).unwrap()); - let partitioned_file = PartitionedFile::new( - "data.parquet", - std::fs::metadata(&parquet_path).unwrap().len(), - ); - let metrics = ExecutionPlanMetricsSet::new(); - let input = CachedMetaReaderFactory::new(object_store).create_liquid_reader( - 0, - partitioned_file, - None, - &metrics, - ); - - let store = test_mount(tmp_dir.path()).await; - let cache = Arc::new( - LiquidCacheParquet::new( - 4, - max_memory_bytes, - max_disk_bytes, - store, - Box::new(LiquidPolicy::new()), - Box::new(Evict), - Box::new(AlwaysHydrate::new()), - ) - .await, - ); - let cached_file = cache.register_or_get_file("data.parquet".to_string(), schema); - let projection = ProjectionMask::roots( - reader_metadata.metadata().file_metadata().schema_descr(), - [0, 1], - ); - let mut builder = LiquidStreamBuilder::new(input, Arc::clone(reader_metadata.metadata())) - .with_row_groups(vec![0, 1]) - .with_projection(projection); - if let Some(row_filter) = row_filter { - builder = builder.with_row_filter(row_filter); - } - let stream = builder.build(cached_file.clone()).unwrap(); - (stream, cache, cached_file, tmp_dir) - } - - async fn collect_liquid_values(stream: LiquidStream) -> (Vec, Vec) { - let batches = stream - .map(|batch| batch.expect("valid liquid stream batch")) - .collect::>() - .await; - let mut a = Vec::new(); - let mut b = Vec::new(); - for batch in batches { - let a_array = batch - .column(0) - .as_any() - .downcast_ref::() - .unwrap(); - let b_array = batch - .column(1) - .as_any() - .downcast_ref::() - .unwrap(); - a.extend(a_array.iter().map(|value| value.unwrap())); - b.extend(b_array.iter().map(|value| value.unwrap())); - } - (a, b) - } - - async fn collect_projected_a(stream: LiquidStream) -> Vec { - let batches = stream - .map(|batch| batch.expect("valid liquid stream batch")) - .collect::>() - .await; - let mut a = Vec::new(); - for batch in batches { - let a_array = batch - .column(0) - .as_any() - .downcast_ref::() - .unwrap(); - a.extend(a_array.iter().map(|value| value.unwrap())); - } - a - } - - fn gt_filter(schema: SchemaRef, literal: i32) -> LiquidRowFilter { - gt_filter_on(schema, "a", 0, literal) - } - - fn gt_filter_on( - schema: SchemaRef, - col_name: &str, - col_idx: usize, - literal: i32, - ) -> LiquidRowFilter { - let expr: Arc = Arc::new(BinaryExpr::new( - Arc::new(Column::new(col_name, col_idx)), - Operator::Gt, - Arc::new(Literal::new(ScalarValue::Int32(Some(literal)))), - )); - let tmp_meta = tempfile::NamedTempFile::new().unwrap(); - write_two_row_group_file(tmp_meta.path(), schema.clone()); - let file = File::open(tmp_meta.path()).unwrap(); - let metadata = ArrowReaderMetadata::load(&file, ArrowReaderOptions::new()).unwrap(); - let builder = FilterCandidateBuilder::new(expr, schema); - let candidate = builder.build(metadata.metadata()).unwrap().unwrap(); - let projection = candidate.projection(metadata.metadata()); - let predicate = LiquidPredicate::try_new(candidate, projection).unwrap(); - LiquidRowFilter::new(vec![predicate]) - } - - async fn make_liquid_stream_with_projection( - max_memory_bytes: usize, - max_disk_bytes: usize, - row_filter: Option, - projection_columns: Vec, - ) -> ( - LiquidStream, - Arc, - CachedFileRef, - tempfile::TempDir, - ) { - let schema = Arc::new(Schema::new(vec![ - Field::new("a", DataType::Int32, false), - Field::new("b", DataType::Int32, false), - ])); - let tmp_dir = tempfile::tempdir().unwrap(); - let parquet_path = tmp_dir.path().join("data.parquet"); - write_two_row_group_file(&parquet_path, schema.clone()); - let metadata_file = File::open(&parquet_path).unwrap(); - let reader_metadata = - ArrowReaderMetadata::load(&metadata_file, ArrowReaderOptions::new()).unwrap(); - let object_store = Arc::new(LocalFileSystem::new_with_prefix(tmp_dir.path()).unwrap()); - let partitioned_file = PartitionedFile::new( - "data.parquet", - std::fs::metadata(&parquet_path).unwrap().len(), - ); - let metrics = ExecutionPlanMetricsSet::new(); - let input = CachedMetaReaderFactory::new(object_store).create_liquid_reader( - 0, - partitioned_file, - None, - &metrics, - ); - - let store = test_mount(tmp_dir.path()).await; - let cache = Arc::new( - LiquidCacheParquet::new( - 4, - max_memory_bytes, - max_disk_bytes, - store, - Box::new(LiquidPolicy::new()), - Box::new(Evict), - Box::new(AlwaysHydrate::new()), - ) - .await, - ); - let cached_file = cache.register_or_get_file("data.parquet".to_string(), schema); - let projection = ProjectionMask::roots( - reader_metadata.metadata().file_metadata().schema_descr(), - projection_columns, - ); - let mut builder = LiquidStreamBuilder::new(input, Arc::clone(reader_metadata.metadata())) - .with_row_groups(vec![0, 1]) - .with_projection(projection); - if let Some(row_filter) = row_filter { - builder = builder.with_row_filter(row_filter); - } - let stream = builder.build(cached_file.clone()).unwrap(); - (stream, cache, cached_file, tmp_dir) - } - - async fn make_single_row_group_stream( - parquet_a: Vec, - projection_columns: Vec, - ) -> (LiquidStream, CachedFileRef, tempfile::TempDir) { - let schema = Arc::new(Schema::new(vec![ - Field::new("a", DataType::Int32, false), - Field::new("b", DataType::Int32, false), - ])); - let tmp_dir = tempfile::tempdir().unwrap(); - let parquet_path = tmp_dir.path().join("data.parquet"); - write_single_row_group_file(&parquet_path, schema.clone(), parquet_a); - let metadata_file = File::open(&parquet_path).unwrap(); - let reader_metadata = - ArrowReaderMetadata::load(&metadata_file, ArrowReaderOptions::new()).unwrap(); - let object_store = Arc::new(LocalFileSystem::new_with_prefix(tmp_dir.path()).unwrap()); - let partitioned_file = PartitionedFile::new( - "data.parquet", - std::fs::metadata(&parquet_path).unwrap().len(), - ); - let metrics = ExecutionPlanMetricsSet::new(); - let input = CachedMetaReaderFactory::new(object_store).create_liquid_reader( - 0, - partitioned_file, - None, - &metrics, - ); - - let store = test_mount(tmp_dir.path()).await; - let cache = Arc::new( - LiquidCacheParquet::new( - 4, - usize::MAX, - usize::MAX, - store, - Box::new(LiquidPolicy::new()), - Box::new(Evict), - Box::new(AlwaysHydrate::new()), - ) - .await, - ); - let cached_file = cache.register_or_get_file("data.parquet".to_string(), schema); - let projection = ProjectionMask::roots( - reader_metadata.metadata().file_metadata().schema_descr(), - projection_columns, - ); - let stream = LiquidStreamBuilder::new(input, Arc::clone(reader_metadata.metadata())) - .with_row_groups(vec![0]) - .with_projection(projection) - .build(cached_file.clone()) - .unwrap(); - (stream, cached_file, tmp_dir) - } - - async fn insert_batches( - row_group: &CachedRowGroupRef, - column_id: usize, - batch_payloads: &[(u16, &[i32])], - ) { - let column = row_group.get_column(column_id as u64).unwrap(); - for (batch_idx, values) in batch_payloads.iter() { - let array: ArrayRef = Arc::new(Int32Array::from(values.to_vec())); - column - .insert(BatchID::from_raw(*batch_idx), array) - .await - .unwrap(); - } - } - - async fn is_cached(row_group: &CachedRowGroupRef, column_id: usize, batch_idx: u16) -> bool { - row_group - .get_column(column_id as u64) - .unwrap() - .get_arrow_array_test_only(BatchID::from_raw(batch_idx)) - .await - .is_some() - } - - #[tokio::test] - async fn cache_full_keeps_inserted_batches_and_skips_failed_inserts() { - let one_array_memory = Arc::new(Int32Array::from(vec![0, 1, 2, 3])).get_array_memory_size(); - let (stream, _cache, cached_file, _tmp_dir) = - make_liquid_stream(one_array_memory * 3, 0, None).await; - - let (a, b) = collect_liquid_values(stream).await; - - assert_eq!(a, vec![0, 1, 2, 3, 4, 5, 6, 7]); - assert_eq!(b, vec![10, 11, 12, 13, 14, 15, 16, 17]); - - let row_group0 = cached_file.create_row_group(0, vec![]); - let row_group1 = cached_file.create_row_group(1, vec![]); - assert!(is_cached(&row_group0, 0, 0).await); - assert!(is_cached(&row_group0, 1, 0).await); - assert!(is_cached(&row_group1, 0, 0).await); - assert!(!is_cached(&row_group1, 1, 0).await); - } - - #[tokio::test] - async fn cache_full_with_row_filter_keeps_lookaside_results_correct() { - let schema = Arc::new(Schema::new(vec![ - Field::new("a", DataType::Int32, false), - Field::new("b", DataType::Int32, false), - ])); - let one_array_memory = Arc::new(Int32Array::from(vec![0, 1, 2, 3])).get_array_memory_size(); - let filter = gt_filter(schema, 2); - let (stream, _cache, cached_file, _tmp_dir) = - make_liquid_stream(one_array_memory * 3, 0, Some(filter)).await; - - let (a, b) = collect_liquid_values(stream).await; - - assert_eq!(a, vec![3, 4, 5, 6, 7]); - assert_eq!(b, vec![13, 14, 15, 16, 17]); - - let row_group0 = cached_file.create_row_group(0, vec![]); - let row_group1 = cached_file.create_row_group(1, vec![]); - assert!(is_cached(&row_group0, 0, 0).await); - assert!(is_cached(&row_group0, 1, 0).await); - assert!(is_cached(&row_group1, 0, 0).await); - assert!(!is_cached(&row_group1, 1, 0).await); - } - - #[tokio::test] - async fn mid_scan_eviction_recovers() { - let (stream, _cache, cached_file, _tmp_dir) = make_liquid_stream(0, 0, None).await; - - let (a, b) = collect_liquid_values(stream).await; - - assert_eq!(a, vec![0, 1, 2, 3, 4, 5, 6, 7]); - assert_eq!(b, vec![10, 11, 12, 13, 14, 15, 16, 17]); - - let row_group0 = cached_file.create_row_group(0, vec![]); - let row_group1 = cached_file.create_row_group(1, vec![]); - assert!(!is_cached(&row_group0, 0, 0).await); - assert!(!is_cached(&row_group0, 1, 0).await); - assert!(!is_cached(&row_group1, 0, 0).await); - assert!(!is_cached(&row_group1, 1, 0).await); - } - - #[tokio::test] - async fn predicate_fallback_uses_predicate_projection() { - let schema = Arc::new(Schema::new(vec![ - Field::new("a", DataType::Int32, false), - Field::new("b", DataType::Int32, false), - ])); - let one_array_memory = Arc::new(Int32Array::from(vec![0, 1, 2, 3])).get_array_memory_size(); - let filter = gt_filter_on(schema, "b", 1, 13); - let (stream, _cache, cached_file, _tmp_dir) = - make_liquid_stream_with_projection(one_array_memory * 3, 0, Some(filter), vec![0]) - .await; - - let a_values = collect_projected_a(stream).await; - - assert_eq!(a_values, vec![4, 5, 6, 7]); - - let row_group0 = cached_file.create_row_group(0, vec![]); - let row_group1 = cached_file.create_row_group(1, vec![]); - assert!(is_cached(&row_group0, 0, 0).await); - assert!(is_cached(&row_group0, 1, 0).await); - assert!(is_cached(&row_group1, 0, 0).await); - assert!(!is_cached(&row_group1, 1, 0).await); - } - - #[tokio::test] - async fn missing_column_falls_back_to_parquet() { - let (stream, _cache, cached_file, _tmp_dir) = - make_liquid_stream(usize::MAX, usize::MAX, None).await; - let row_group0 = cached_file.create_row_group(0, vec![]); - let row_group1 = cached_file.create_row_group(1, vec![]); - insert_batches(&row_group0, 0, &[(0, &[0, 1, 2, 3])]).await; - insert_batches(&row_group1, 0, &[(0, &[4, 5, 6, 7])]).await; - - let (a, b) = collect_liquid_values(stream).await; - - assert_eq!(a, vec![0, 1, 2, 3, 4, 5, 6, 7]); - assert_eq!(b, vec![10, 11, 12, 13, 14, 15, 16, 17]); - assert!(is_cached(&row_group0, 1, 0).await); - assert!(is_cached(&row_group1, 1, 0).await); - } - - #[tokio::test] - async fn fallback_stream_advances_across_misses() { - let parquet_a = vec![ - 100, 101, 102, 103, 4, 5, 6, 7, 200, 201, 202, 203, 12, 13, 14, 15, - ]; - let (stream, cached_file, _tmp_dir) = - make_single_row_group_stream(parquet_a, vec![0]).await; - let row_group = cached_file.create_row_group(0, vec![]); - insert_batches(&row_group, 0, &[(0, &[0, 1, 2, 3]), (2, &[8, 9, 10, 11])]).await; - - let a_values = collect_projected_a(stream).await; - - assert_eq!(a_values, (0..16).collect::>()); - assert!(is_cached(&row_group, 0, 0).await); - assert!(is_cached(&row_group, 0, 1).await); - assert!(is_cached(&row_group, 0, 2).await); - assert!(is_cached(&row_group, 0, 3).await); - } - - /// Build a stream over a single row group with the given cache batch size. - async fn make_single_row_group_stream_with_cache_batch_size( - row_count: i32, - cache_batch_size: usize, - limit: Option, - ) -> (LiquidStream, Arc, tempfile::TempDir) { - let schema = Arc::new(Schema::new(vec![ - Field::new("a", DataType::Int32, false), - Field::new("b", DataType::Int32, false), - ])); - let tmp_dir = tempfile::tempdir().unwrap(); - let parquet_path = tmp_dir.path().join("data.parquet"); - write_single_row_group_file(&parquet_path, schema.clone(), (0..row_count).collect()); - let metadata_file = File::open(&parquet_path).unwrap(); - let reader_metadata = - ArrowReaderMetadata::load(&metadata_file, ArrowReaderOptions::new()).unwrap(); - let object_store = Arc::new(LocalFileSystem::new_with_prefix(tmp_dir.path()).unwrap()); - let partitioned_file = PartitionedFile::new( - "data.parquet", - std::fs::metadata(&parquet_path).unwrap().len(), - ); - let metrics = ExecutionPlanMetricsSet::new(); - let input = CachedMetaReaderFactory::new(object_store).create_liquid_reader( - 0, - partitioned_file, - None, - &metrics, - ); - - let store = test_mount(tmp_dir.path()).await; - let cache = Arc::new( - LiquidCacheParquet::new( - cache_batch_size, - 1024 * 1024 * 1024, - 1024 * 1024 * 1024, - store, - Box::new(LiquidPolicy::new()), - Box::new(Evict), - Box::new(AlwaysHydrate::new()), - ) - .await, - ); - let cached_file = cache.register_or_get_file("data.parquet".to_string(), schema); - let projection = ProjectionMask::roots( - reader_metadata.metadata().file_metadata().schema_descr(), - [0], - ); - let stream = LiquidStreamBuilder::new(input, Arc::clone(reader_metadata.metadata())) - .with_row_groups(vec![0]) - .with_limit(limit) - .with_projection(projection) - .build(cached_file) - .unwrap(); - (stream, cache, tmp_dir) - } - - async fn try_collect_a(stream: LiquidStream) -> Result, ParquetError> { - let batches: Vec<_> = stream.collect().await; - let mut a = Vec::new(); - for batch in batches { - let batch = batch?; - let array = batch - .column(0) - .as_any() - .downcast_ref::() - .unwrap(); - a.extend(array.iter().map(|value| value.unwrap())); - } - Ok(a) - } - - /// A decode error ends the scan, and the stream stays ended. - /// - /// This load-bears for correctness rather than tidiness. On an error the - /// reader returns `ProcessResult::Emit(Err(..))` without incrementing - /// `current_batch_id` and leaves itself in `ReaderState::Ready`, so a scan - /// that resumed would read the next selection window against a stale batch - /// id — the misalignment this module exists to prevent. The file here has two - /// row groups, so a stream that failed to terminate would carry on into the - /// second one instead of stopping. - #[tokio::test] - async fn decode_error_ends_the_stream() { - let (stream, _cache, _cached_file, tmp_dir) = - make_liquid_stream(1024 * 1024, 1024 * 1024, None).await; - - // Truncate the file now that the metadata is cached, so the parquet - // fallback's byte reads fail on the first batch. - File::create(tmp_dir.path().join("data.parquet")).unwrap(); - - let mut stream = Box::pin(stream); - assert!( - matches!(stream.next().await, Some(Err(_))), - "expected the truncated file to surface a decode error" - ); - assert!( - stream.next().await.is_none(), - "stream must not resume into the next row group after an error" - ); - assert!( - stream.next().await.is_none(), - "stream must stay ended once it has ended" - ); - } - - /// Issue #13: the reader indexes the cache by batch id and the parquet - /// fallback turns that id back into rows with the cache batch size, so the - /// scan must read in cache-sized batches. `build` takes the size from the - /// cache for exactly that reason — assert the emitted batches follow it and - /// the rows stay in order, including when the size does not divide the row - /// group evenly. - #[tokio::test] - async fn stream_reads_in_cache_sized_batches() { - let (stream, _cache, _tmp) = - make_single_row_group_stream_with_cache_batch_size(20, 8, None).await; - let batches: Vec<_> = stream - .map(|batch| batch.expect("valid liquid stream batch")) - .collect() - .await; - - assert_eq!( - batches.iter().map(|b| b.num_rows()).collect::>(), - vec![8, 8, 4] - ); - let values: Vec = batches - .iter() - .flat_map(|batch| { - let array = batch - .column(0) - .as_any() - .downcast_ref::() - .unwrap(); - array.iter().map(|value| value.unwrap()).collect::>() - }) - .collect(); - assert_eq!(values, (0..20).collect::>()); - } - - /// A limit that stops the scan mid-row-group still yields aligned rows. This - /// is the shape that used to return rows from the wrong offsets without any - /// error at all. - #[tokio::test] - async fn limited_scan_stays_aligned() { - let (stream, _cache, _tmp) = - make_single_row_group_stream_with_cache_batch_size(32, 8, Some(12)).await; - let values = try_collect_a(stream).await.expect("stream should succeed"); - assert_eq!(values, (0..12).collect::>()); - } -} diff --git a/src/datafusion/src/reader/runtime/mod.rs b/src/datafusion/src/reader/runtime/mod.rs index bfde28dcc..b647962fa 100644 --- a/src/datafusion/src/reader/runtime/mod.rs +++ b/src/datafusion/src/reader/runtime/mod.rs @@ -1,7 +1,11 @@ +pub(crate) use liquid_cache_reader::apply_predicates; pub(crate) use liquid_predicate::extract_multi_column_or; -pub(crate) use liquid_stream::LiquidStreamBuilder; +pub(crate) use morsel::{ + LiquidRowGroupPlanner, LiquidRowGroupPrefetchContext, build_projection_schema, +}; +pub(crate) use utils::{get_root_column_ids, take_next_batch}; mod liquid_cache_reader; mod liquid_predicate; -mod liquid_stream; +mod morsel; mod utils; diff --git a/src/datafusion/src/reader/runtime/morsel.rs b/src/datafusion/src/reader/runtime/morsel.rs new file mode 100644 index 000000000..1c6931ffb --- /dev/null +++ b/src/datafusion/src/reader/runtime/morsel.rs @@ -0,0 +1,249 @@ +use std::{fmt, pin::Pin, sync::Arc}; + +use arrow::array::{RecordBatch, RecordBatchOptions}; +use arrow_schema::{Schema, SchemaRef}; +use datafusion::{error::DataFusionError, physical_expr::projection::Projector}; +use datafusion_datasource::morsel::Morsel; +use futures::{Stream, StreamExt, stream::BoxStream}; +use parquet::{ + arrow::{ + ProjectionMask, + arrow_reader::{ArrowPredicate, RowSelection, RowSelector}, + }, + file::metadata::ParquetMetaData, +}; + +use crate::{ + cache::{CachedFileRef, CachedRowGroupRef, LiquidCacheParquetRef, RowGroupSnapshots}, + reader::plantime::{LiquidFileMetrics, LiquidRowFilter, ParquetMetadataCacheReader}, +}; + +use super::{ + liquid_cache_reader::{ + LiquidCacheReader, LiquidCacheReaderConfig, ParquetFallback, ParquetFallbackConfig, + }, + utils::get_root_column_ids, +}; + +pub(crate) struct LiquidRowGroupPlanner { + pub(crate) metadata: Arc, + pub(crate) input: ParquetMetadataCacheReader, + pub(crate) row_filter: Option, + pub(crate) cached_file: CachedFileRef, + pub(crate) projection: ProjectionMask, + pub(crate) batch_size: usize, + pub(crate) stream_schema: SchemaRef, + pub(crate) output_schema: SchemaRef, + pub(crate) projector: Arc, + pub(crate) replace_schema: bool, + pub(crate) span: Option>, + pub(crate) liquid_cache: LiquidCacheParquetRef, + pub(crate) metrics: LiquidFileMetrics, +} + +impl LiquidRowGroupPlanner { + fn cache_details(&self) -> CacheDetails { + let schema_descr = self.metadata.file_metadata().schema_descr(); + let mut predicate_projection: Option = None; + if let Some(filter) = &self.row_filter { + for predicate in filter.predicates() { + let projection = predicate.projection(); + if let Some(predicate_projection) = &mut predicate_projection { + predicate_projection.union(projection); + } else { + predicate_projection = Some(projection.clone()); + } + } + } + let mut cache_projection = self.projection.clone(); + if let Some(predicate_projection) = &predicate_projection { + cache_projection.union(predicate_projection); + } + CacheDetails { + projection_column_ids: get_root_column_ids(schema_descr, &self.projection), + cache_column_ids: get_root_column_ids(schema_descr, &cache_projection), + predicate_column_ids: predicate_projection + .as_ref() + .map(|projection| get_root_column_ids(schema_descr, projection)) + .unwrap_or_default(), + cache_projection, + } + } + + fn fallback_config( + &self, + row_group_idx: usize, + details: &CacheDetails, + cache_batch_size: usize, + ) -> ParquetFallbackConfig { + ParquetFallbackConfig { + row_group_idx, + metadata: Arc::clone(&self.metadata), + input: self.input.clone(), + cache_projection: details.cache_projection.clone(), + cache_column_ids: details.cache_column_ids.clone(), + cache_batch_size, + row_count: self.metadata.row_group(row_group_idx).num_rows() as usize, + } + } + + pub(crate) fn estimated_bytes(&self, row_group_idx: usize) -> usize { + let details = self.cache_details(); + self.metadata + .row_group(row_group_idx) + .columns() + .iter() + .enumerate() + .filter(|(idx, _)| details.cache_projection.leaf_included(*idx)) + .map(|(_, column)| column.uncompressed_size() as usize) + .sum() + } + + pub(crate) fn prefetch_context( + &self, + row_group_idx: usize, + snapshots: Arc, + ) -> LiquidRowGroupPrefetchContext { + let details = self.cache_details(); + let cached_row_group = self.cached_file.create_row_group_with_snapshots( + row_group_idx as u64, + details.predicate_column_ids.clone(), + snapshots, + ); + let fallback = ParquetFallback::new(self.fallback_config( + row_group_idx, + &details, + cached_row_group.batch_size(), + )); + LiquidRowGroupPrefetchContext { + cached_row_group, + cache_column_ids: details.cache_column_ids, + predicate_column_ids: details.predicate_column_ids, + projection_column_ids: details.projection_column_ids, + row_filter: self.row_filter.clone(), + fallback, + } + } + + pub(crate) fn plan( + &self, + row_group_idx: usize, + selection: Option, + snapshots: Arc, + ) -> Option { + let metadata = self.metadata.row_group(row_group_idx); + let details = self.cache_details(); + + let selection = selection + .unwrap_or_else(|| vec![RowSelector::select(metadata.num_rows() as usize)].into()); + if selection.row_count() == 0 { + return None; + } + + let schema_descr = self.metadata.file_metadata().schema_descr(); + let projection_columns = get_root_column_ids(schema_descr, &self.projection); + let cached_row_group = self.cached_file.create_row_group_with_snapshots( + row_group_idx as u64, + details.predicate_column_ids.clone(), + snapshots, + ); + let cache_batch_size = cached_row_group.batch_size(); + + Some(LiquidRowGroupMorsel { + config: LiquidCacheReaderConfig { + batch_size: self.batch_size, + selection, + row_filter: self.row_filter.clone(), + cached_row_group, + projection_columns, + schema: Arc::clone(&self.stream_schema), + parquet_fallback: self.fallback_config(row_group_idx, &details, cache_batch_size), + }, + output_schema: Arc::clone(&self.output_schema), + projector: Arc::clone(&self.projector), + replace_schema: self.replace_schema, + span: self.span.clone(), + }) + } +} + +struct CacheDetails { + cache_projection: ProjectionMask, + cache_column_ids: Vec, + predicate_column_ids: Vec, + projection_column_ids: Vec, +} + +pub(crate) struct LiquidRowGroupPrefetchContext { + pub(crate) cached_row_group: CachedRowGroupRef, + pub(crate) cache_column_ids: Vec, + pub(crate) predicate_column_ids: Vec, + pub(crate) projection_column_ids: Vec, + pub(crate) row_filter: Option, + pub(crate) fallback: ParquetFallback, +} + +pub(crate) fn build_projection_schema( + file_schema: &SchemaRef, + projection_column_ids: &[usize], +) -> SchemaRef { + let fields = projection_column_ids + .iter() + .filter_map(|column_id| file_schema.fields().get(*column_id)) + .map(|field| field.as_ref().clone()) + .collect::>(); + Arc::new(Schema::new(fields)) +} + +pub(crate) struct LiquidRowGroupMorsel { + config: LiquidCacheReaderConfig, + output_schema: SchemaRef, + projector: Arc, + replace_schema: bool, + span: Option>, +} + +impl fmt::Debug for LiquidRowGroupMorsel { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.debug_struct("LiquidRowGroupMorsel") + .finish_non_exhaustive() + } +} + +impl Morsel for LiquidRowGroupMorsel { + fn into_stream(self: Box) -> BoxStream<'static, datafusion::error::Result> { + let Self { + config, + output_schema, + projector, + replace_schema, + span, + } = *self; + let mut reader = LiquidCacheReader::new(config); + let stream = futures::stream::poll_fn(move |cx| { + let _guard = span.as_ref().map(|span| span.set_local_parent()); + Pin::new(&mut reader).poll_next(cx) + }); + + stream + .map(|batch| batch.map_err(|error| DataFusionError::External(Box::new(error)))) + .map(move |batch| { + batch.and_then(|batch| { + let batch = projector.project_batch(&batch)?; + if replace_schema { + let (_schema, arrays, num_rows) = batch.into_parts(); + let options = RecordBatchOptions::new().with_row_count(Some(num_rows)); + RecordBatch::try_new_with_options( + Arc::clone(&output_schema), + arrays, + &options, + ) + .map_err(Into::into) + } else { + Ok(batch) + } + }) + }) + .boxed() + } +} diff --git a/src/datafusion/src/reader/runtime/utils.rs b/src/datafusion/src/reader/runtime/utils.rs index a05b79942..ffa8b363c 100644 --- a/src/datafusion/src/reader/runtime/utils.rs +++ b/src/datafusion/src/reader/runtime/utils.rs @@ -1,10 +1,7 @@ use std::collections::VecDeque; use parquet::{ - arrow::{ - ProjectionMask, - arrow_reader::{RowSelection, RowSelector}, - }, + arrow::{ProjectionMask, arrow_reader::RowSelector}, schema::types::SchemaDescriptor, }; @@ -28,67 +25,9 @@ pub(crate) fn get_root_column_ids( .collect() } -pub(crate) fn offset_row_selection(selection: RowSelection, offset: usize) -> RowSelection { - if offset == 0 { - return selection; - } - - let mut selected_count = 0; - let mut skipped_count = 0; - - let mut selectors: Vec = selection.into(); - - let find = selectors.iter().position(|selector| match selector.skip { - true => { - skipped_count += selector.row_count; - false - } - false => { - selected_count += selector.row_count; - selected_count > offset - } - }); - - let split_idx = match find { - Some(idx) => idx, - None => { - selectors.clear(); - return RowSelection::from(selectors); - } - }; - - let mut new_selectors = Vec::with_capacity(selectors.len() - split_idx + 1); - new_selectors.push(RowSelector::skip(skipped_count + offset)); - new_selectors.push(RowSelector::select(selected_count - offset)); - new_selectors.extend_from_slice(&selectors[split_idx + 1..]); - - RowSelection::from(new_selectors) -} - -pub(crate) fn limit_row_selection(selection: RowSelection, mut limit: usize) -> RowSelection { - let mut selectors: Vec = selection.into(); - - if limit == 0 { - selectors.clear(); - } - - for (idx, selection) in selectors.iter_mut().enumerate() { - if !selection.skip { - if selection.row_count >= limit { - selection.row_count = limit; - selectors.truncate(idx + 1); - break; - } else { - limit -= selection.row_count; - } - } - } - RowSelection::from(selectors) -} - /// Take the next batch from the selection queue. /// The returning selection will have exactly the batch size, or less if the selection is exhausted. -pub(super) fn take_next_batch( +pub(crate) fn take_next_batch( selection: &mut VecDeque, batch_size: usize, ) -> Option> {