Compare commits
168 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| efb1e652be | |||
| 800463c1b0 | |||
| fa1143b11b | |||
| d4f0331fb0 | |||
| 16b34f5eec | |||
| f30739b908 | |||
| 19ca421550 | |||
| 11dcdb589d | |||
| ffb39a8137 | |||
| 193a00a628 | |||
| 1db2b7de72 | |||
| a9be747c00 | |||
| 9a0deb39c3 | |||
| ee070aaecd | |||
| 6645a68ebc | |||
| e5cdf922fa | |||
| 97c1812668 | |||
| 398e6697bd | |||
| 4693adfe02 | |||
| d56006d26f | |||
| a5a2781bca | |||
| 1104a8deaf | |||
| a1efd5e011 | |||
| 39f67c0cf9 | |||
| b3e112de15 | |||
| 8cb1f4768b | |||
| bea54cca10 | |||
| ff817190d3 | |||
| e5662c1bf6 | |||
| f7daab7642 | |||
| ed0e84d4fc | |||
| e8dd50ce5a | |||
| 51d0583145 | |||
| bb4f337a78 | |||
| 4c5c92ac38 | |||
| 2f1e51e262 | |||
| 52b1e86b63 | |||
| 2e2a7a0ab9 | |||
| b94f90f923 | |||
| 362ba12517 | |||
| f401061fa6 | |||
| 3a52fc7c76 | |||
| c262af6923 | |||
| f73106cece | |||
| 24a4a7ae75 | |||
| 1bcbce2bc6 | |||
| 573a783d2f | |||
| bcd3c76285 | |||
| ca7bf03cfc | |||
| 37bda06c0c | |||
| 287d7b75a9 | |||
| 0f5ead1ab5 | |||
| 84ba575a3f | |||
| bca71efb32 | |||
| b23a0308eb | |||
| 2a03538e11 | |||
| e00b41df50 | |||
| 4b13a91aa5 | |||
| 5b0098a072 | |||
| 1b79eecb58 | |||
| 71c33e63b2 | |||
| acd2f0ce2f | |||
| 8298260c64 | |||
| 24d71a468a | |||
| a677d645fd | |||
| a24cf55c2b | |||
| 464f9d8b85 | |||
| 743ab60f48 | |||
| 4daf8bbc50 | |||
| 2201faee5f | |||
| 7bad9f2484 | |||
| 91545f832c | |||
| 8f23da638b | |||
| cfea91ce88 | |||
| cc9ce2501a | |||
| 2465c2cad6 | |||
| 59702f3c2e | |||
| e2ce56ae51 | |||
| ed4c60833b | |||
| cebacb3c35 | |||
| 689e31284a | |||
| c450a8d346 | |||
| 77881a1c92 | |||
| 7de55be63d | |||
| 2606dce6bc | |||
| 8912a1abdb | |||
| d25ea7562b | |||
| ea1f9802d7 | |||
| 07e90e5898 | |||
| 0fd47ab72d | |||
| 7979004d77 | |||
| 210bc9e1c7 | |||
| 205f9d6db9 | |||
| 406bf2531c | |||
| 42a12051d8 | |||
| 477a8a2c96 | |||
| 9a6ee3d18a | |||
| 29585a4aa3 | |||
| 1c2c0633cd | |||
| 5158ba64b8 | |||
| 51e0ef7a64 | |||
| cd0efe50d3 | |||
| 541c3f5722 | |||
| 59d626dacd | |||
| 29e6480ea0 | |||
| db4cd2825c | |||
| b1c4f8e7d7 | |||
| feaae7de28 | |||
| c2d76b5466 | |||
| d4b9d16073 | |||
| 094a35e435 | |||
| 697481fd29 | |||
| bb6f32a2ad | |||
| 335394f4b5 | |||
| 538c593b7b | |||
| 2dc8b30d92 | |||
| d54b17ba81 | |||
| cbf260d082 | |||
| 371d54a478 | |||
| e83b137be5 | |||
| eecb795a0e | |||
| 7dc1f86d71 | |||
| 38712e2607 | |||
| eb577ea4f3 | |||
| dd56a3a8a8 | |||
| aee5fbb8c4 | |||
| 85130958bd | |||
| a3f3e6a265 | |||
| b005c0a790 | |||
| 2ba8415039 | |||
| b1be35cc5f | |||
| 9ff925e31e | |||
| e64d2e2a55 | |||
| 2b6ace888d | |||
| 4b83987ea1 | |||
| c6cb66b597 | |||
| 64861f8142 | |||
| eb5f2b3648 | |||
| fcdf4a9172 | |||
| 1001eb8b5e | |||
| f0ce2dd50d | |||
| 1c2a1c1204 | |||
| 66bdfff454 | |||
| 1e50f8df80 | |||
| 736c41ecd6 | |||
| ac8df4c9e4 | |||
| e3e534f4ad | |||
| 764e3ecf18 | |||
| 10a27f9678 | |||
| 66e712e066 | |||
| c5a3f89c5e | |||
| 29f81a141f | |||
| d2e3993398 | |||
| a411100fa4 | |||
| 0a993d7a30 | |||
| 7c2d46f1e2 | |||
| c3ed9fb17b | |||
| 0f40c66eb7 | |||
| 7298978bcb | |||
| 8f69f987a4 | |||
| 3e9b8655b7 | |||
| 727935ede6 | |||
| 9a2612b1b5 | |||
| 7c0d999144 | |||
| 81d98f6b9a | |||
| 46e87e5928 | |||
| 9bc928db65 | |||
| 434344f6e9 |
+3
-8
@@ -10,12 +10,7 @@ rustflags = ["-C", "target-feature=-crt-static"]
|
||||
[target.aarch64-unknown-linux-musl]
|
||||
rustflags = ["-C", "target-feature=-crt-static"]
|
||||
|
||||
# Android/Termux: no hardcoded linker so native Termux builds use the system cc.
|
||||
# For CI cross-compilation the linker is set via CARGO_TARGET_AARCH64_LINUX_ANDROID_LINKER env var.
|
||||
[target.aarch64-linux-android]
|
||||
rustflags = [
|
||||
"-C",
|
||||
"linker=aarch64-linux-android-clang",
|
||||
"-C",
|
||||
"link-args=-rdynamic",
|
||||
"-C",
|
||||
"default-linker-libraries",
|
||||
]
|
||||
rustflags = ["-C", "link-args=-rdynamic"]
|
||||
|
||||
@@ -0,0 +1,195 @@
|
||||
name: e2e Tests
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
paths-ignore:
|
||||
- '**.md'
|
||||
- 'doc/**'
|
||||
pull_request:
|
||||
branches: [main]
|
||||
paths-ignore:
|
||||
- '**.md'
|
||||
- 'doc/**'
|
||||
|
||||
env:
|
||||
CARGO_TERM_COLOR: always
|
||||
MACOSX_DEPLOYMENT_TARGET: "13"
|
||||
# Force Node 24 for all JS-based actions to avoid the libuv
|
||||
# process_title assertion crash on Windows (known Node 20 bug).
|
||||
FORCE_JAVASCRIPT_ACTIONS_TO_NODE24: true
|
||||
|
||||
jobs:
|
||||
lua-tests:
|
||||
name: e2e (${{ matrix.os }})
|
||||
runs-on: ${{ matrix.os }}
|
||||
# e2e tests could be flaky on CI so we do not block release creation if they failed
|
||||
continue-on-error: ${{ github.ref == 'refs/heads/main' && github.event_name == 'push' }}
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
include:
|
||||
- os: ubuntu-latest
|
||||
- os: macos-latest
|
||||
- os: windows-latest
|
||||
target: x86_64-pc-windows-msvc
|
||||
steps:
|
||||
- uses: actions/checkout@v5
|
||||
- uses: oven-sh/setup-bun@v2
|
||||
- uses: actions/setup-node@v6
|
||||
- name: Install Zig
|
||||
uses: mlugg/setup-zig@v2
|
||||
with:
|
||||
version: 0.16.0
|
||||
|
||||
- name: Install Rust
|
||||
uses: actions-rust-lang/setup-rust-toolchain@v1.15.4
|
||||
with:
|
||||
cache: true
|
||||
cache-on-failure: false
|
||||
cache-key: "v2-lua-e2e"
|
||||
rustflags: ""
|
||||
target: ${{ matrix.target || '' }}
|
||||
|
||||
- name: Build Rust binary (Windows)
|
||||
if: matrix.target
|
||||
run: cargo build --release --target ${{ matrix.target }} -p fff-nvim --features zlob
|
||||
|
||||
- name: Copy binary to target/release (Windows)
|
||||
if: matrix.target
|
||||
shell: bash
|
||||
run: |
|
||||
cp target/${{ matrix.target }}/release/fff_nvim.dll target/release/fff_nvim.dll
|
||||
|
||||
- name: Verify Windows DLL has no unexpected dependencies
|
||||
if: matrix.target
|
||||
shell: pwsh
|
||||
run: |
|
||||
# Find dumpbin via vswhere (always available on GitHub Actions Windows runners)
|
||||
$vsPath = & "${env:ProgramFiles(x86)}\Microsoft Visual Studio\Installer\vswhere.exe" -latest -property installationPath
|
||||
$dumpbin = Get-ChildItem "$vsPath" -Recurse -Filter "dumpbin.exe" | Select-Object -First 1
|
||||
if (-not $dumpbin) { Write-Error "dumpbin.exe not found"; exit 1 }
|
||||
|
||||
$deps = & $dumpbin.FullName /DEPENDENTS target\release\fff_nvim.dll | Out-String
|
||||
Write-Host $deps
|
||||
# zlob must be statically linked - fail if zlob.dll appears as a dependency
|
||||
if ($deps -match 'zlob\.dll') {
|
||||
Write-Error "fff_nvim.dll has unexpected dynamic dependency on zlob.dll - zlob should be statically linked"
|
||||
exit 1
|
||||
}
|
||||
|
||||
- name: Build Rust binary
|
||||
if: ${{ !matrix.target }}
|
||||
run: cargo build --release -p fff-nvim --features zlob
|
||||
|
||||
- name: Install Neovim
|
||||
uses: rhysd/action-setup-vim@v1
|
||||
with:
|
||||
neovim: true
|
||||
version: v0.10.4
|
||||
|
||||
- name: Clone plenary.nvim
|
||||
shell: bash
|
||||
run: git clone --depth 1 https://github.com/nvim-lua/plenary.nvim ../plenary.nvim
|
||||
|
||||
- name: Run Lua tests
|
||||
shell: bash
|
||||
run: make test-lua
|
||||
|
||||
- name: Run version resolution tests
|
||||
shell: bash
|
||||
run: make test-version
|
||||
|
||||
- name: Run bun tests
|
||||
shell: bash
|
||||
if: ${{ matrix.os != 'windows-latest' }}
|
||||
run: make test-bun
|
||||
|
||||
- name: Install Node.js
|
||||
if: ${{ matrix.os != 'ubuntu-latest' }}
|
||||
uses: actions/setup-node@v6
|
||||
with:
|
||||
node-version: "25"
|
||||
|
||||
- name: Install node dependencies
|
||||
shell: bash
|
||||
run: cd packages/fff-node && npm install
|
||||
|
||||
- name: Run node tests
|
||||
shell: bash
|
||||
run: make test-node
|
||||
|
||||
# Regression for https://github.com/dmtrKovalenko/fff/issues/480: build &
|
||||
# run @ff-labs/fff-node end-to-end on real Alpine Linux (musl). Forces
|
||||
# findBinary() through the npm-package resolver so detectLinuxLibc()
|
||||
# actually runs.
|
||||
alpine-musl:
|
||||
name: e2e (alpine-musl)
|
||||
runs-on: ubuntu-latest
|
||||
container: node:22-alpine
|
||||
continue-on-error: ${{ github.ref == 'refs/heads/main' && github.event_name == 'push' }}
|
||||
defaults:
|
||||
run:
|
||||
shell: sh
|
||||
steps:
|
||||
- name: Install build deps
|
||||
run: apk add --no-cache git rust cargo musl-dev
|
||||
|
||||
- uses: actions/checkout@v5
|
||||
|
||||
# libgit2 refuses repos owned by a different user; checkout in a
|
||||
# container can land at a uid mismatch, so opt every dir in.
|
||||
- name: Mark workspace safe for git
|
||||
run: git config --global --add safe.directory '*'
|
||||
|
||||
- name: Sanity check libc is musl
|
||||
run: |
|
||||
if ! ldd --version 2>&1 | grep -qi musl; then
|
||||
echo "FAIL: container is not running musl libc"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
- uses: actions/cache@v4
|
||||
with:
|
||||
path: |
|
||||
~/.cargo/registry
|
||||
~/.cargo/git
|
||||
target
|
||||
key: alpine-musl-cargo-${{ hashFiles('**/Cargo.lock') }}
|
||||
restore-keys: |
|
||||
alpine-musl-cargo-
|
||||
|
||||
- name: Build libfff_c (musl)
|
||||
run: cargo build --release -p fff-c
|
||||
|
||||
- name: Install workspace npm deps
|
||||
run: npm install --no-package-lock
|
||||
|
||||
# Upstream @yuuang/ffi-rs-linux-x64-musl ships with libc:"glibc" in
|
||||
# its package.json (a publishing bug in ffi-rs), so npm filters it
|
||||
# out. Force-install it so the FFI runtime is present on Alpine.
|
||||
- name: Install ffi-rs musl runtime
|
||||
run: |
|
||||
FFI_RS_VERSION=$(node -p "require('ffi-rs/package.json').version")
|
||||
npm install --no-package-lock --no-save --force \
|
||||
"@yuuang/ffi-rs-linux-x64-musl@${FFI_RS_VERSION}"
|
||||
|
||||
# Stage the freshly built libfff_c.so as the platform npm package
|
||||
# so findBinary() resolves through the @ff-labs/fff-bin-* path —
|
||||
# this is what exercises detectLinuxLibc().
|
||||
- name: Stage musl bin package
|
||||
run: |
|
||||
PKG_DIR=node_modules/@ff-labs/fff-bin-linux-x64-musl
|
||||
mkdir -p "$PKG_DIR"
|
||||
cp target/release/libfff_c.so "$PKG_DIR/libfff_c.so"
|
||||
cat >"$PKG_DIR/package.json" <<'JSON'
|
||||
{ "name": "@ff-labs/fff-bin-linux-x64-musl", "version": "0.0.0" }
|
||||
JSON
|
||||
|
||||
- name: Build fff-node
|
||||
working-directory: packages/fff-node
|
||||
run: npm run build
|
||||
|
||||
- name: Run fff-node e2e suite
|
||||
working-directory: packages/fff-node
|
||||
run: node test/e2e.mjs
|
||||
@@ -1,87 +0,0 @@
|
||||
name: Lua E2E Tests
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
pull_request:
|
||||
branches: [main]
|
||||
|
||||
env:
|
||||
CARGO_TERM_COLOR: always
|
||||
MACOSX_DEPLOYMENT_TARGET: "13"
|
||||
|
||||
jobs:
|
||||
lua-tests:
|
||||
name: Lua E2E (${{ matrix.os }})
|
||||
runs-on: ${{ matrix.os }}
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
include:
|
||||
- os: ubuntu-latest
|
||||
- os: macos-latest
|
||||
- os: windows-latest
|
||||
target: x86_64-pc-windows-msvc
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Install Zig
|
||||
uses: mlugg/setup-zig@v2
|
||||
with:
|
||||
version: 0.15.2
|
||||
|
||||
- name: Install Rust
|
||||
uses: actions-rust-lang/setup-rust-toolchain@v1
|
||||
with:
|
||||
cache: true
|
||||
cache-on-failure: true
|
||||
cache-key: "v1-lua-e2e"
|
||||
rustflags: ""
|
||||
target: ${{ matrix.target || '' }}
|
||||
|
||||
- name: Build Rust binary (Windows)
|
||||
if: matrix.target
|
||||
run: cargo build --release --target ${{ matrix.target }} -p fff-nvim
|
||||
|
||||
- name: Copy binary to target/release (Windows)
|
||||
if: matrix.target
|
||||
shell: bash
|
||||
run: |
|
||||
cp target/${{ matrix.target }}/release/fff_nvim.dll target/release/fff_nvim.dll
|
||||
|
||||
- name: Verify Windows DLL has no unexpected dependencies
|
||||
if: matrix.target
|
||||
shell: pwsh
|
||||
run: |
|
||||
# Find dumpbin via vswhere (always available on GitHub Actions Windows runners)
|
||||
$vsPath = & "${env:ProgramFiles(x86)}\Microsoft Visual Studio\Installer\vswhere.exe" -latest -property installationPath
|
||||
$dumpbin = Get-ChildItem "$vsPath" -Recurse -Filter "dumpbin.exe" | Select-Object -First 1
|
||||
if (-not $dumpbin) { Write-Error "dumpbin.exe not found"; exit 1 }
|
||||
|
||||
$deps = & $dumpbin.FullName /DEPENDENTS target\release\fff_nvim.dll | Out-String
|
||||
Write-Host $deps
|
||||
# zlob must be statically linked - fail if zlob.dll appears as a dependency
|
||||
if ($deps -match 'zlob\.dll') {
|
||||
Write-Error "fff_nvim.dll has unexpected dynamic dependency on zlob.dll - zlob should be statically linked"
|
||||
exit 1
|
||||
}
|
||||
|
||||
- name: Build Rust binary
|
||||
if: ${{ !matrix.target }}
|
||||
run: cargo build --release -p fff-nvim
|
||||
|
||||
- name: Install Neovim
|
||||
uses: rhysd/action-setup-vim@v1
|
||||
with:
|
||||
neovim: true
|
||||
version: v0.10.4
|
||||
|
||||
- name: Clone plenary.nvim
|
||||
shell: bash
|
||||
run: git clone --depth 1 https://github.com/nvim-lua/plenary.nvim ../plenary.nvim
|
||||
|
||||
- name: Run Lua tests
|
||||
shell: bash
|
||||
run: |
|
||||
nvim --headless -u tests/minimal_init.lua \
|
||||
-c "PlenaryBustedFile tests/fff_core_spec.lua" 2>&1
|
||||
@@ -0,0 +1,56 @@
|
||||
name: Lua CI
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
paths-ignore:
|
||||
- '**.md'
|
||||
- 'doc/**'
|
||||
pull_request:
|
||||
branches: [main]
|
||||
paths-ignore:
|
||||
- '**.md'
|
||||
- 'doc/**'
|
||||
|
||||
jobs:
|
||||
lua-ls:
|
||||
name: lua-language-server type check
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v5
|
||||
|
||||
- name: Install Neovim
|
||||
run: |
|
||||
curl -L https://github.com/neovim/neovim/releases/download/v0.11.5/nvim-linux-x86_64.tar.gz -o /opt/nvim.tar.gz
|
||||
mkdir /opt/nvim
|
||||
tar xzf /opt/nvim.tar.gz -C /opt/nvim
|
||||
mv /opt/nvim/nvim-linux-x86_64/* /opt/nvim
|
||||
echo "/opt/nvim/bin" >> $GITHUB_PATH
|
||||
|
||||
- name: Install lua-language-server
|
||||
run: |
|
||||
curl -L "https://github.com/LuaLS/lua-language-server/releases/download/3.17.1/lua-language-server-3.17.1-linux-x64.tar.gz" -o /opt/lls.tar.gz
|
||||
mkdir /opt/lls
|
||||
tar -xzf /opt/lls.tar.gz -C /opt/lls
|
||||
echo "/opt/lls/bin" >> $GITHUB_PATH
|
||||
|
||||
- name: Clone snacks.nvim
|
||||
run: git clone --depth=1 https://github.com/folke/snacks.nvim /opt/snacks.nvim
|
||||
|
||||
- name: Run lua-language-server
|
||||
run: lua-language-server --configpath .luarc.ci.json --check=.
|
||||
|
||||
luacheck:
|
||||
name: luacheck lint
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v5
|
||||
|
||||
- name: Install luacheck
|
||||
run: |
|
||||
sudo apt-get update -qq
|
||||
sudo apt-get install -y luarocks
|
||||
sudo luarocks install luacheck
|
||||
|
||||
- name: Run luacheck
|
||||
run: luacheck lua/
|
||||
@@ -3,8 +3,14 @@ name: Nix CI
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
paths-ignore:
|
||||
- '**.md'
|
||||
- 'doc/**'
|
||||
pull_request:
|
||||
branches: [main]
|
||||
paths-ignore:
|
||||
- '**.md'
|
||||
- 'doc/**'
|
||||
|
||||
jobs:
|
||||
check:
|
||||
@@ -13,7 +19,7 @@ jobs:
|
||||
id-token: "write"
|
||||
contents: "read"
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@v5
|
||||
- uses: DeterminateSystems/nix-installer-action@main
|
||||
- uses: DeterminateSystems/magic-nix-cache-action@main
|
||||
- uses: DeterminateSystems/flake-checker-action@main
|
||||
|
||||
@@ -1,7 +1,8 @@
|
||||
on:
|
||||
push:
|
||||
branches-ignore:
|
||||
- main
|
||||
schedule:
|
||||
- cron: "0 4 * * *"
|
||||
workflow_dispatch:
|
||||
|
||||
name: docs
|
||||
|
||||
jobs:
|
||||
@@ -9,27 +10,64 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: write
|
||||
pull-requests: write
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@v5
|
||||
with:
|
||||
# fetch last 2 commits required for auto force push back
|
||||
ref: main
|
||||
fetch-depth: 2
|
||||
|
||||
- name: Extract Neovim section from README.md
|
||||
run: |
|
||||
awk '
|
||||
/^<details id="neovim-plugin">/ { capture=1; next }
|
||||
capture && /^<\/details>/ { capture=0; exit }
|
||||
capture && /^<summary>$/ { next }
|
||||
capture && /^<\/summary>$/ { next }
|
||||
capture && /<h2>.*<\/h2>/ {
|
||||
gsub(/<\/?h2>/, "")
|
||||
sub(/^[[:space:]]+/, "")
|
||||
print "# " $0
|
||||
print ""
|
||||
print "The best file search picker for Neovim. Frecency-ranked, typo-resistant, git-award, very fast."
|
||||
print ""
|
||||
next
|
||||
}
|
||||
capture { print }
|
||||
' README.md > .panvimdoc-input.md
|
||||
test -s .panvimdoc-input.md
|
||||
|
||||
- name: panvimdoc
|
||||
uses: kdheepak/panvimdoc@main
|
||||
with:
|
||||
vimdoc: fff.nvim
|
||||
pandoc: .panvimdoc-input.md
|
||||
version: "Neovim >= 0.10.0"
|
||||
demojify: true
|
||||
treesitter: true
|
||||
|
||||
- name: Get last commit message
|
||||
id: last-commit
|
||||
run: |
|
||||
echo "message=$(git log -1 --pretty=%s)" >> $GITHUB_OUTPUT
|
||||
echo "author=$(git log -1 --pretty=\"%an <%ae>\")" >> $GITHUB_OUTPUT
|
||||
- name: Cleanup intermediate file
|
||||
run: rm -f .panvimdoc-input.md
|
||||
|
||||
- uses: stefanzweifel/git-auto-commit-action@v6
|
||||
- name: Create pull request
|
||||
id: cpr
|
||||
uses: peter-evans/create-pull-request@v7
|
||||
with:
|
||||
commit_author: ${{ steps.last-commit.outputs.author }}
|
||||
commit_message: "chore: Update docs for - ${{ steps.last-commit.outputs.message }}"
|
||||
branch: bot/regenerate-vimdoc
|
||||
token: ${{ secrets.GUSTAV_PAT }}
|
||||
delete-branch: true
|
||||
title: "chore: regenerate Neovim vimdoc"
|
||||
commit-message: |
|
||||
chore: regenerate Neovim vimdoc
|
||||
|
||||
Co-authored-by: Dmitriy Kovalenko <dmtr.kovalenko@outlook.com>
|
||||
author: "gustav-fff <66k7bxj9m6@privaterelay.appleid.com>"
|
||||
committer: "gustav-fff <66k7bxj9m6@privaterelay.appleid.com>"
|
||||
body: Automated vimdoc regeneration from README.md, scribed by Gustav.
|
||||
add-paths: doc/fff.nvim.txt
|
||||
|
||||
- name: Enable auto-merge
|
||||
if: steps.cpr.outputs.pull-request-number
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.GUSTAV_PAT }}
|
||||
run: gh pr merge --auto --squash "${{ steps.cpr.outputs.pull-request-number }}"
|
||||
|
||||
+362
-92
@@ -2,9 +2,14 @@ name: Prebuild
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
branches: [main, fix/download-version]
|
||||
tags:
|
||||
- "v*"
|
||||
pull_request:
|
||||
|
||||
env:
|
||||
FORCE_JAVASCRIPT_ACTIONS_TO_NODE24: true
|
||||
|
||||
jobs:
|
||||
build-nvim:
|
||||
name: Build Neovim ${{ matrix.target }}
|
||||
@@ -15,48 +20,56 @@ jobs:
|
||||
matrix:
|
||||
include:
|
||||
## Linux builds (using cargo-zigbuild)
|
||||
# Glibc 2.17 (RHEL 7, CentOS 7 compatible)
|
||||
# Glibc 2.31 (Ubuntu 20.04, Debian 11, RHEL 9).
|
||||
# Rust 1.91+ requires glibc >= 2.31 for std::sys::random::getrandom,
|
||||
# copy_file_range, and statx; earlier targets (2.17) no longer link.
|
||||
- os: ubuntu-latest
|
||||
target: x86_64-unknown-linux-gnu
|
||||
zigbuild_target: x86_64-unknown-linux-gnu.2.17
|
||||
artifact_name: target/x86_64-unknown-linux-gnu/release/libfff_nvim.so
|
||||
zigbuild_target: x86_64-unknown-linux-gnu.2.31
|
||||
artifact_name: target/x86_64-unknown-linux-gnu/ci/libfff_nvim.so
|
||||
ext: so
|
||||
- os: ubuntu-latest
|
||||
target: aarch64-unknown-linux-gnu
|
||||
zigbuild_target: aarch64-unknown-linux-gnu.2.17
|
||||
artifact_name: target/aarch64-unknown-linux-gnu/release/libfff_nvim.so
|
||||
zigbuild_target: aarch64-unknown-linux-gnu.2.31
|
||||
artifact_name: target/aarch64-unknown-linux-gnu/ci/libfff_nvim.so
|
||||
ext: so
|
||||
# Musl (statically linked)
|
||||
- os: ubuntu-latest
|
||||
target: x86_64-unknown-linux-musl
|
||||
artifact_name: target/x86_64-unknown-linux-musl/release/libfff_nvim.so
|
||||
artifact_name: target/x86_64-unknown-linux-musl/ci/libfff_nvim.so
|
||||
ext: so
|
||||
- os: ubuntu-latest
|
||||
target: aarch64-unknown-linux-musl
|
||||
artifact_name: target/aarch64-unknown-linux-musl/release/libfff_nvim.so
|
||||
artifact_name: target/aarch64-unknown-linux-musl/ci/libfff_nvim.so
|
||||
ext: so
|
||||
|
||||
## Android (Termux)
|
||||
- os: ubuntu-latest
|
||||
target: aarch64-linux-android
|
||||
artifact_name: target/aarch64-linux-android/ci/libfff_nvim.so
|
||||
ext: so
|
||||
|
||||
## macOS builds
|
||||
- os: macos-latest
|
||||
target: x86_64-apple-darwin
|
||||
artifact_name: target/x86_64-apple-darwin/release/libfff_nvim.dylib
|
||||
artifact_name: target/x86_64-apple-darwin/ci/libfff_nvim.dylib
|
||||
ext: dylib
|
||||
- os: macos-latest
|
||||
target: aarch64-apple-darwin
|
||||
artifact_name: target/aarch64-apple-darwin/release/libfff_nvim.dylib
|
||||
artifact_name: target/aarch64-apple-darwin/ci/libfff_nvim.dylib
|
||||
ext: dylib
|
||||
|
||||
- os: windows-latest
|
||||
target: x86_64-pc-windows-msvc
|
||||
artifact_name: target/x86_64-pc-windows-msvc/release/fff_nvim.dll
|
||||
artifact_name: target/x86_64-pc-windows-msvc/ci/fff_nvim.dll
|
||||
ext: dll
|
||||
- os: windows-latest
|
||||
target: aarch64-pc-windows-msvc
|
||||
artifact_name: target/aarch64-pc-windows-msvc/release/fff_nvim.dll
|
||||
artifact_name: target/aarch64-pc-windows-msvc/ci/fff_nvim.dll
|
||||
ext: dll
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@v5
|
||||
with:
|
||||
persist-credentials: false
|
||||
|
||||
@@ -66,29 +79,47 @@ jobs:
|
||||
- name: Install Zig
|
||||
uses: mlugg/setup-zig@v2
|
||||
with:
|
||||
version: 0.15.2
|
||||
version: 0.16.0
|
||||
|
||||
- name: Install cargo-zigbuild
|
||||
if: contains(matrix.os, 'ubuntu')
|
||||
run: cargo install cargo-zigbuild
|
||||
|
||||
- name: Build for Linux
|
||||
if: contains(matrix.os, 'ubuntu')
|
||||
if: contains(matrix.os, 'ubuntu') && !contains(matrix.target, 'android')
|
||||
run: |
|
||||
cargo zigbuild --release --target ${{ matrix.zigbuild_target || matrix.target }} -p fff-nvim
|
||||
cargo zigbuild --profile ci --target ${{ matrix.zigbuild_target || matrix.target }} -p fff-nvim --features zlob
|
||||
mv "${{ matrix.artifact_name }}" "${{ matrix.target }}.${{ matrix.ext }}"
|
||||
|
||||
- name: Build for Android (Termux)
|
||||
if: contains(matrix.target, 'android')
|
||||
run: |
|
||||
NDK_BIN="$ANDROID_NDK/toolchains/llvm/prebuilt/linux-x86_64/bin"
|
||||
|
||||
# NDK clang for C deps (libgit2, lmdb, blake3) that need Bionic sysroot headers
|
||||
export CC_aarch64_linux_android="$NDK_BIN/aarch64-linux-android24-clang"
|
||||
export CXX_aarch64_linux_android="$NDK_BIN/aarch64-linux-android24-clang++"
|
||||
export AR_aarch64_linux_android="$NDK_BIN/llvm-ar"
|
||||
export CARGO_TARGET_AARCH64_LINUX_ANDROID_LINKER="$NDK_BIN/aarch64-linux-android24-clang"
|
||||
|
||||
cargo build --profile ci --target ${{ matrix.target }} -p fff-nvim --features zlob
|
||||
mv "${{ matrix.artifact_name }}" "${{ matrix.target }}.${{ matrix.ext }}"
|
||||
|
||||
- name: Build for macOS
|
||||
if: contains(matrix.os, 'macos')
|
||||
run: |
|
||||
MACOSX_DEPLOYMENT_TARGET="13" cargo build --release --target ${{ matrix.target }} -p fff-nvim
|
||||
MACOSX_DEPLOYMENT_TARGET="13" cargo build --profile ci --target ${{ matrix.target }} -p fff-nvim --features zlob
|
||||
mv "${{ matrix.artifact_name }}" "${{ matrix.target }}.${{ matrix.ext }}"
|
||||
|
||||
- name: Ad-hoc sign macOS binary
|
||||
if: contains(matrix.os, 'macos')
|
||||
run: codesign --force --sign - "${{ matrix.target }}.${{ matrix.ext }}"
|
||||
|
||||
- name: Build for Windows
|
||||
if: contains(matrix.os, 'windows')
|
||||
shell: bash
|
||||
run: |
|
||||
cargo build --release --target ${{ matrix.target }} -p fff-nvim
|
||||
cargo build --profile ci --target ${{ matrix.target }} -p fff-nvim --features zlob
|
||||
mv "${{ matrix.artifact_name }}" "${{ matrix.target }}.${{ matrix.ext }}"
|
||||
|
||||
- name: Upload artifacts
|
||||
@@ -108,45 +139,68 @@ jobs:
|
||||
## Linux builds
|
||||
- os: ubuntu-latest
|
||||
target: x86_64-unknown-linux-gnu
|
||||
zigbuild_target: x86_64-unknown-linux-gnu.2.17
|
||||
artifact_name: target/x86_64-unknown-linux-gnu/release/libfff_c.so
|
||||
zigbuild_target: x86_64-unknown-linux-gnu.2.31
|
||||
artifact_name: target/x86_64-unknown-linux-gnu/ci/libfff_c.so
|
||||
npm_package: fff-bin-linux-x64-gnu
|
||||
lib_filename: libfff_c.so
|
||||
ext: so
|
||||
- os: ubuntu-latest
|
||||
target: aarch64-unknown-linux-gnu
|
||||
zigbuild_target: aarch64-unknown-linux-gnu.2.17
|
||||
artifact_name: target/aarch64-unknown-linux-gnu/release/libfff_c.so
|
||||
zigbuild_target: aarch64-unknown-linux-gnu.2.31
|
||||
artifact_name: target/aarch64-unknown-linux-gnu/ci/libfff_c.so
|
||||
npm_package: fff-bin-linux-arm64-gnu
|
||||
lib_filename: libfff_c.so
|
||||
ext: so
|
||||
- os: ubuntu-latest
|
||||
target: x86_64-unknown-linux-musl
|
||||
artifact_name: target/x86_64-unknown-linux-musl/release/libfff_c.so
|
||||
artifact_name: target/x86_64-unknown-linux-musl/ci/libfff_c.so
|
||||
npm_package: fff-bin-linux-x64-musl
|
||||
lib_filename: libfff_c.so
|
||||
ext: so
|
||||
- os: ubuntu-latest
|
||||
target: aarch64-unknown-linux-musl
|
||||
artifact_name: target/aarch64-unknown-linux-musl/release/libfff_c.so
|
||||
artifact_name: target/aarch64-unknown-linux-musl/ci/libfff_c.so
|
||||
npm_package: fff-bin-linux-arm64-musl
|
||||
lib_filename: libfff_c.so
|
||||
ext: so
|
||||
|
||||
## Android (Termux)
|
||||
- os: ubuntu-latest
|
||||
target: aarch64-linux-android
|
||||
artifact_name: target/aarch64-linux-android/ci/libfff_c.so
|
||||
lib_filename: libfff_c.so
|
||||
ext: so
|
||||
|
||||
## macOS builds
|
||||
- os: macos-latest
|
||||
target: x86_64-apple-darwin
|
||||
artifact_name: target/x86_64-apple-darwin/release/libfff_c.dylib
|
||||
artifact_name: target/x86_64-apple-darwin/ci/libfff_c.dylib
|
||||
npm_package: fff-bin-darwin-x64
|
||||
lib_filename: libfff_c.dylib
|
||||
ext: dylib
|
||||
- os: macos-latest
|
||||
target: aarch64-apple-darwin
|
||||
artifact_name: target/aarch64-apple-darwin/release/libfff_c.dylib
|
||||
artifact_name: target/aarch64-apple-darwin/ci/libfff_c.dylib
|
||||
npm_package: fff-bin-darwin-arm64
|
||||
lib_filename: libfff_c.dylib
|
||||
ext: dylib
|
||||
|
||||
## Windows builds
|
||||
- os: windows-latest
|
||||
target: x86_64-pc-windows-msvc
|
||||
artifact_name: target/x86_64-pc-windows-msvc/release/fff_c.dll
|
||||
artifact_name: target/x86_64-pc-windows-msvc/ci/fff_c.dll
|
||||
npm_package: fff-bin-win32-x64
|
||||
lib_filename: fff_c.dll
|
||||
ext: dll
|
||||
- os: windows-latest
|
||||
target: aarch64-pc-windows-msvc
|
||||
artifact_name: target/aarch64-pc-windows-msvc/release/fff_c.dll
|
||||
artifact_name: target/aarch64-pc-windows-msvc/ci/fff_c.dll
|
||||
npm_package: fff-bin-win32-arm64
|
||||
lib_filename: fff_c.dll
|
||||
ext: dll
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@v5
|
||||
with:
|
||||
persist-credentials: false
|
||||
|
||||
@@ -156,7 +210,120 @@ jobs:
|
||||
- name: Install Zig
|
||||
uses: mlugg/setup-zig@v2
|
||||
with:
|
||||
version: 0.15.2
|
||||
version: 0.16.0
|
||||
|
||||
- name: Install cargo-zigbuild
|
||||
if: contains(matrix.os, 'ubuntu')
|
||||
run: cargo install cargo-zigbuild
|
||||
|
||||
- name: Build for Linux
|
||||
if: contains(matrix.os, 'ubuntu') && !contains(matrix.target, 'android')
|
||||
run: |
|
||||
cargo zigbuild --profile ci --target ${{ matrix.zigbuild_target || matrix.target }} -p fff-c --features zlob
|
||||
mv "${{ matrix.artifact_name }}" "c-lib-${{ matrix.target }}.${{ matrix.ext }}"
|
||||
|
||||
- name: Build for Android (Termux)
|
||||
if: contains(matrix.target, 'android')
|
||||
run: |
|
||||
NDK_BIN="$ANDROID_NDK/toolchains/llvm/prebuilt/linux-x86_64/bin"
|
||||
|
||||
export CC_aarch64_linux_android="$NDK_BIN/aarch64-linux-android24-clang"
|
||||
export CXX_aarch64_linux_android="$NDK_BIN/aarch64-linux-android24-clang++"
|
||||
export AR_aarch64_linux_android="$NDK_BIN/llvm-ar"
|
||||
export CARGO_TARGET_AARCH64_LINUX_ANDROID_LINKER="$NDK_BIN/aarch64-linux-android24-clang"
|
||||
|
||||
cargo build --profile ci --target ${{ matrix.target }} -p fff-c --features zlob
|
||||
mv "${{ matrix.artifact_name }}" "c-lib-${{ matrix.target }}.${{ matrix.ext }}"
|
||||
|
||||
- name: Build for macOS
|
||||
if: contains(matrix.os, 'macos')
|
||||
run: |
|
||||
MACOSX_DEPLOYMENT_TARGET="13" cargo build --profile ci --target ${{ matrix.target }} -p fff-c --features zlob
|
||||
mv "${{ matrix.artifact_name }}" "c-lib-${{ matrix.target }}.${{ matrix.ext }}"
|
||||
|
||||
- name: Ad-hoc sign macOS binary
|
||||
if: contains(matrix.os, 'macos')
|
||||
run: codesign --force --sign - "c-lib-${{ matrix.target }}.${{ matrix.ext }}"
|
||||
|
||||
- name: Build for Windows
|
||||
if: contains(matrix.os, 'windows')
|
||||
shell: bash
|
||||
run: |
|
||||
cargo build --profile ci --target ${{ matrix.target }} -p fff-c --features zlob
|
||||
mv "${{ matrix.artifact_name }}" "c-lib-${{ matrix.target }}.${{ matrix.ext }}"
|
||||
|
||||
- name: Prepare npm package
|
||||
if: "!contains(matrix.target, 'android')"
|
||||
shell: bash
|
||||
run: |
|
||||
# Copy the built binary into the platform npm package directory
|
||||
cp "c-lib-${{ matrix.target }}.${{ matrix.ext }}" "packages/${{ matrix.npm_package }}/${{ matrix.lib_filename }}"
|
||||
|
||||
- name: Upload C library artifact
|
||||
uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: c-lib-${{ matrix.target }}
|
||||
path: c-lib-${{ matrix.target }}.*
|
||||
|
||||
- name: Upload npm package artifact
|
||||
if: "!contains(matrix.target, 'android')"
|
||||
uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: npm-${{ matrix.npm_package }}
|
||||
path: packages/${{ matrix.npm_package }}/
|
||||
|
||||
build-mcp:
|
||||
name: Build MCP ${{ matrix.target }}
|
||||
runs-on: ${{ matrix.os }}
|
||||
permissions:
|
||||
contents: read
|
||||
strategy:
|
||||
matrix:
|
||||
include:
|
||||
## Linux builds (using cargo-zigbuild)
|
||||
- os: ubuntu-latest
|
||||
target: x86_64-unknown-linux-gnu
|
||||
zigbuild_target: x86_64-unknown-linux-gnu.2.31
|
||||
artifact_name: target/x86_64-unknown-linux-gnu/ci/fff-mcp
|
||||
- os: ubuntu-latest
|
||||
target: aarch64-unknown-linux-gnu
|
||||
zigbuild_target: aarch64-unknown-linux-gnu.2.31
|
||||
artifact_name: target/aarch64-unknown-linux-gnu/ci/fff-mcp
|
||||
- os: ubuntu-latest
|
||||
target: x86_64-unknown-linux-musl
|
||||
artifact_name: target/x86_64-unknown-linux-musl/ci/fff-mcp
|
||||
- os: ubuntu-latest
|
||||
target: aarch64-unknown-linux-musl
|
||||
artifact_name: target/aarch64-unknown-linux-musl/ci/fff-mcp
|
||||
|
||||
## macOS builds
|
||||
- os: macos-latest
|
||||
target: x86_64-apple-darwin
|
||||
artifact_name: target/x86_64-apple-darwin/ci/fff-mcp
|
||||
- os: macos-latest
|
||||
target: aarch64-apple-darwin
|
||||
artifact_name: target/aarch64-apple-darwin/ci/fff-mcp
|
||||
|
||||
## Windows builds
|
||||
- os: windows-latest
|
||||
target: x86_64-pc-windows-msvc
|
||||
artifact_name: target/x86_64-pc-windows-msvc/ci/fff-mcp.exe
|
||||
- os: windows-latest
|
||||
target: aarch64-pc-windows-msvc
|
||||
artifact_name: target/aarch64-pc-windows-msvc/ci/fff-mcp.exe
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v5
|
||||
with:
|
||||
persist-credentials: false
|
||||
|
||||
- name: Install Rust
|
||||
run: rustup target add ${{ matrix.target }}
|
||||
|
||||
- name: Install Zig
|
||||
uses: mlugg/setup-zig@v2
|
||||
with:
|
||||
version: 0.16.0
|
||||
|
||||
- name: Install cargo-zigbuild
|
||||
if: contains(matrix.os, 'ubuntu')
|
||||
@@ -165,36 +332,44 @@ jobs:
|
||||
- name: Build for Linux
|
||||
if: contains(matrix.os, 'ubuntu')
|
||||
run: |
|
||||
cargo zigbuild --release --target ${{ matrix.zigbuild_target || matrix.target }} -p fff-c
|
||||
mv "${{ matrix.artifact_name }}" "c-lib-${{ matrix.target }}.${{ matrix.ext }}"
|
||||
cargo zigbuild --profile ci --target ${{ matrix.zigbuild_target || matrix.target }} -p fff-mcp --features zlob
|
||||
cp "${{ matrix.artifact_name }}" "fff-mcp-${{ matrix.target }}"
|
||||
|
||||
- name: Build for macOS
|
||||
if: contains(matrix.os, 'macos')
|
||||
run: |
|
||||
MACOSX_DEPLOYMENT_TARGET="13" cargo build --release --target ${{ matrix.target }} -p fff-c
|
||||
mv "${{ matrix.artifact_name }}" "c-lib-${{ matrix.target }}.${{ matrix.ext }}"
|
||||
MACOSX_DEPLOYMENT_TARGET="13" cargo build --profile ci --target ${{ matrix.target }} -p fff-mcp --features zlob
|
||||
cp "${{ matrix.artifact_name }}" "fff-mcp-${{ matrix.target }}"
|
||||
|
||||
- name: Ad-hoc sign macOS binary
|
||||
if: contains(matrix.os, 'macos')
|
||||
run: codesign --force --sign - "fff-mcp-${{ matrix.target }}"
|
||||
|
||||
- name: Build for Windows
|
||||
if: contains(matrix.os, 'windows')
|
||||
shell: bash
|
||||
run: |
|
||||
cargo build --release --target ${{ matrix.target }} -p fff-c
|
||||
mv "${{ matrix.artifact_name }}" "c-lib-${{ matrix.target }}.${{ matrix.ext }}"
|
||||
cargo build --profile ci --target ${{ matrix.target }} -p fff-mcp --features zlob
|
||||
cp "${{ matrix.artifact_name }}" "fff-mcp-${{ matrix.target }}.exe"
|
||||
|
||||
- name: Upload artifacts
|
||||
- name: Upload artifact
|
||||
uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: c-lib-${{ matrix.target }}
|
||||
path: c-lib-${{ matrix.target }}.*
|
||||
name: mcp-${{ matrix.target }}
|
||||
path: fff-mcp-${{ matrix.target }}*
|
||||
|
||||
release:
|
||||
name: Release
|
||||
needs: [build-nvim, build-c]
|
||||
needs: [build-nvim, build-c, build-mcp]
|
||||
runs-on: ubuntu-latest
|
||||
if: github.event_name == 'push' && (github.ref == 'refs/heads/main' || github.ref == 'refs/heads/fix/download-version' || startsWith(github.ref, 'refs/tags/v'))
|
||||
permissions:
|
||||
contents: write
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@v5
|
||||
|
||||
- name: Install Lua
|
||||
uses: leafo/gh-actions-lua@v12
|
||||
|
||||
- name: Download artifacts
|
||||
uses: actions/download-artifact@v4
|
||||
@@ -231,6 +406,24 @@ jobs:
|
||||
rmdir "$dir" 2>/dev/null || true
|
||||
done
|
||||
|
||||
- name: Flatten MCP artifacts
|
||||
working-directory: ./binaries
|
||||
run: |
|
||||
for dir in mcp-*/; do
|
||||
for file in "$dir"*; do
|
||||
if [ -f "$file" ]; then
|
||||
filename=$(basename "$file")
|
||||
mv "$file" "./$filename"
|
||||
fi
|
||||
done
|
||||
rmdir "$dir" 2>/dev/null || true
|
||||
done
|
||||
|
||||
- name: Remove npm package artifacts from release binaries
|
||||
working-directory: ./binaries
|
||||
run: |
|
||||
rm -rf npm-*
|
||||
|
||||
- name: Generate checksums
|
||||
working-directory: ./binaries
|
||||
run: |
|
||||
@@ -241,25 +434,22 @@ jobs:
|
||||
fi
|
||||
done
|
||||
|
||||
- name: Prepare tag
|
||||
id: vars
|
||||
shell: bash
|
||||
run: |
|
||||
sha="$(git rev-parse --short HEAD)"
|
||||
echo "tag=$sha" >> $GITHUB_OUTPUT
|
||||
- name: Determine version
|
||||
id: version
|
||||
run: lua scripts/determine-version.lua
|
||||
|
||||
- name: Upload Release Assets
|
||||
uses: softprops/action-gh-release@v2
|
||||
with:
|
||||
name: "${{ steps.vars.outputs.tag }}"
|
||||
tag_name: "${{ steps.vars.outputs.tag }}"
|
||||
name: "${{ steps.version.outputs.version }}"
|
||||
tag_name: "${{ steps.version.outputs.is_release == 'true' && format('v{0}', steps.version.outputs.version) || steps.version.outputs.version }}"
|
||||
token: ${{ github.token }}
|
||||
files: ./binaries/*
|
||||
draft: false
|
||||
prerelease: true
|
||||
generate_release_notes: false
|
||||
prerelease: ${{ steps.version.outputs.is_release != 'true' }}
|
||||
generate_release_notes: ${{ steps.version.outputs.is_release == 'true' }}
|
||||
body: |
|
||||
Nightly release from commit: ${{ github.sha }}
|
||||
${{ steps.version.outputs.is_release == 'true' && format('Release {0}', steps.version.outputs.version) || format('Nightly release from commit: {0}', github.sha) }}
|
||||
|
||||
## Neovim Plugin
|
||||
- `{target}.so` / `.dylib` / `.dll` - Lua module for Neovim
|
||||
@@ -267,47 +457,127 @@ jobs:
|
||||
## C FFI Library (for Bun/Node/Python)
|
||||
- `c-lib-{target}.so` / `.dylib` / `.dll` - C FFI library
|
||||
|
||||
comment-on-pr:
|
||||
name: Comment on PR
|
||||
needs: [build-nvim, build-c]
|
||||
runs-on: ubuntu-latest
|
||||
if: github.event_name == 'pull_request'
|
||||
permissions:
|
||||
pull-requests: write
|
||||
steps:
|
||||
- name: Get short SHA
|
||||
id: vars
|
||||
run: echo "short_sha=${GITHUB_SHA::7}" >> $GITHUB_OUTPUT
|
||||
## MCP Server
|
||||
- `fff-mcp-{target}` - MCP server binary
|
||||
|
||||
- name: Find existing comment
|
||||
uses: peter-evans/find-comment@v3
|
||||
id: find-comment
|
||||
with:
|
||||
issue-number: ${{ github.event.pull_request.number }}
|
||||
comment-author: "github-actions[bot]"
|
||||
body-includes: "<!-- fff-nvim-build-comment -->"
|
||||
|
||||
- name: Create or update PR comment
|
||||
uses: peter-evans/create-or-update-comment@v4
|
||||
with:
|
||||
comment-id: ${{ steps.find-comment.outputs.comment-id }}
|
||||
issue-number: ${{ github.event.pull_request.number }}
|
||||
edit-mode: replace
|
||||
body: |
|
||||
<!-- fff-nvim-build-comment -->
|
||||
## Build Artifacts for your PR
|
||||
|
||||
### Neovim Plugin
|
||||
Test with lazy.nvim:
|
||||
```lua
|
||||
{
|
||||
"dmtrKovalenko/fff.nvim",
|
||||
tag = "${{ steps.vars.outputs.short_sha }}",
|
||||
}
|
||||
Install with:
|
||||
```sh
|
||||
curl -fsSL https://raw.githubusercontent.com/dmtrKovalenko/fff.nvim/main/install-mcp.sh | bash
|
||||
```
|
||||
|
||||
### Bun/TypeScript Package
|
||||
The `fff` npm package will download binaries from this release automatically.
|
||||
crates-publish:
|
||||
name: Publish Rust crates
|
||||
needs: [build-nvim, build-c, build-mcp]
|
||||
runs-on: ubuntu-latest
|
||||
if: github.event_name == 'push' && (github.ref == 'refs/heads/main' || github.ref == 'refs/heads/fix/download-version' || startsWith(github.ref, 'refs/tags/v'))
|
||||
|
||||
---
|
||||
*Built from ${{ github.sha }}*
|
||||
steps:
|
||||
- uses: actions/checkout@v5
|
||||
|
||||
- name: Install Lua
|
||||
uses: leafo/gh-actions-lua@v12
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
|
||||
- name: Install cargo-edit
|
||||
run: cargo install cargo-edit
|
||||
|
||||
- name: Determine version
|
||||
id: version
|
||||
run: lua scripts/determine-version.lua
|
||||
|
||||
- name: Publish crates
|
||||
env:
|
||||
CARGO_REGISTRY_TOKEN: ${{ secrets.CARGO_REGISTRY_TOKEN }}
|
||||
run: make publish-crates V="${{ steps.version.outputs.version }}"
|
||||
|
||||
npm-publish:
|
||||
name: Publish npm packages
|
||||
needs: [build-c]
|
||||
runs-on: ubuntu-latest
|
||||
if: github.event_name == 'push' && (github.ref == 'refs/heads/main' || github.ref == 'refs/heads/fix/download-version' || startsWith(github.ref, 'refs/tags/v'))
|
||||
permissions:
|
||||
contents: read
|
||||
steps:
|
||||
- uses: actions/checkout@v5
|
||||
|
||||
- name: Install Lua
|
||||
uses: leafo/gh-actions-lua@v12
|
||||
|
||||
- name: Setup Node.js
|
||||
uses: actions/setup-node@v4
|
||||
with:
|
||||
node-version: "25"
|
||||
registry-url: "https://registry.npmjs.org"
|
||||
|
||||
- name: Determine version
|
||||
id: version
|
||||
run: lua scripts/determine-version.lua
|
||||
|
||||
- name: Download npm package artifacts
|
||||
uses: actions/download-artifact@v4
|
||||
with:
|
||||
pattern: npm-*
|
||||
path: ./npm-packages
|
||||
|
||||
- name: Publish platform packages
|
||||
env:
|
||||
NODE_AUTH_TOKEN: ${{ secrets.NPM_TOKEN }}
|
||||
run: |
|
||||
VERSION="${{ steps.version.outputs.version }}"
|
||||
TAG="${{ steps.version.outputs.npm_tag }}"
|
||||
|
||||
for pkg_dir in ./npm-packages/npm-*/; do
|
||||
if [ -d "$pkg_dir" ]; then
|
||||
pkg_name=$(node -p "require('${pkg_dir}package.json').name")
|
||||
echo "Publishing ${pkg_name}@${VERSION} with tag ${TAG}..."
|
||||
|
||||
make set-npm-version PKG="$pkg_dir" VERSION="$VERSION"
|
||||
|
||||
cd "$pkg_dir"
|
||||
npm publish --tag "$TAG" --access public || echo "Failed to publish ${pkg_name} (may already exist)"
|
||||
cd -
|
||||
fi
|
||||
done
|
||||
|
||||
- name: Publish bun package
|
||||
env:
|
||||
NODE_AUTH_TOKEN: ${{ secrets.NPM_TOKEN }}
|
||||
run: |
|
||||
VERSION="${{ steps.version.outputs.version }}"
|
||||
TAG="${{ steps.version.outputs.npm_tag }}"
|
||||
|
||||
echo "Publishing @ff-labs/fff-bun@${VERSION} with tag ${TAG}..."
|
||||
make set-npm-version PKG=packages/fff-bun VERSION="$VERSION"
|
||||
|
||||
cd packages/fff-bun
|
||||
npm publish --tag "$TAG" --access public || echo "Failed to publish @ff-labs/fff-bun (may already exist)"
|
||||
|
||||
- name: Publish Node.js package
|
||||
env:
|
||||
NODE_AUTH_TOKEN: ${{ secrets.NPM_TOKEN }}
|
||||
run: |
|
||||
VERSION="${{ steps.version.outputs.version }}"
|
||||
TAG="${{ steps.version.outputs.npm_tag }}"
|
||||
|
||||
echo "Publishing @ff-labs/fff-node@${VERSION} with tag ${TAG}..."
|
||||
make set-npm-version PKG=packages/fff-node VERSION="$VERSION"
|
||||
|
||||
cd packages/fff-node
|
||||
npm install
|
||||
npm run build
|
||||
npm publish --tag "$TAG" --access public || echo "Failed to publish @ff-labs/fff-node (may already exist)"
|
||||
|
||||
- name: Publish pi-fff package
|
||||
env:
|
||||
NODE_AUTH_TOKEN: ${{ secrets.NPM_TOKEN }}
|
||||
run: |
|
||||
VERSION="${{ steps.version.outputs.version }}"
|
||||
TAG="${{ steps.version.outputs.npm_tag }}"
|
||||
|
||||
echo "Publishing @ff-labs/pi-fff@${VERSION} with tag ${TAG}..."
|
||||
make set-npm-version PKG=packages/pi-fff VERSION="$VERSION"
|
||||
|
||||
cd packages/pi-fff
|
||||
npm publish --tag "$TAG" --access public || echo "Failed to publish @ff-labs/pi-fff (may already exist)"
|
||||
|
||||
+72
-10
@@ -3,8 +3,14 @@ name: Rust CI
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
paths-ignore:
|
||||
- '**.md'
|
||||
- 'doc/**'
|
||||
pull_request:
|
||||
branches: [main]
|
||||
paths-ignore:
|
||||
- '**.md'
|
||||
- 'doc/**'
|
||||
|
||||
env:
|
||||
CARGO_TERM_COLOR: always
|
||||
@@ -17,19 +23,23 @@ jobs:
|
||||
name: Test
|
||||
runs-on: ${{ matrix.os }}
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
os: [ubuntu-latest, macos-latest]
|
||||
os: [ubuntu-latest, macos-latest, windows-latest]
|
||||
# Guard against deadlocks in the shared-picker / watcher teardown
|
||||
# path: a stuck test would otherwise consume a full 6h CI slot.
|
||||
timeout-minutes: 15
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@v5
|
||||
|
||||
# Zig is required to compile zlob
|
||||
- name: Install Zig
|
||||
uses: mlugg/setup-zig@v2
|
||||
uses: goto-bus-stop/setup-zig@v2
|
||||
with:
|
||||
version: 0.15.2
|
||||
version: 0.16.0
|
||||
|
||||
- name: Install Rust
|
||||
uses: actions-rust-lang/setup-rust-toolchain@v1
|
||||
uses: actions-rust-lang/setup-rust-toolchain@v1.15.4
|
||||
with:
|
||||
cache: true
|
||||
cache-on-failure: true
|
||||
@@ -37,13 +47,65 @@ jobs:
|
||||
components: rustfmt, clippy
|
||||
|
||||
- name: Run tests
|
||||
run: cargo test --verbose --workspace --exclude fff-nvim
|
||||
run: cargo test --features zlob --workspace --exclude fff-nvim
|
||||
|
||||
stress-test:
|
||||
name: Stress Test (Watcher + Git)
|
||||
runs-on: ${{ matrix.os }}
|
||||
strategy:
|
||||
# Keep going after one OS fails so we can see whether a bug
|
||||
# reproduces everywhere or is platform-specific.
|
||||
fail-fast: false
|
||||
matrix:
|
||||
os: [ubuntu-latest, macos-latest, windows-latest]
|
||||
# Long-running; don't let a stuck watcher thread burn a full CI
|
||||
# timeout. Two scenarios should finish well under this limit.
|
||||
timeout-minutes: 20
|
||||
steps:
|
||||
- uses: actions/checkout@v5
|
||||
|
||||
- name: Install Zig
|
||||
uses: goto-bus-stop/setup-zig@v2
|
||||
with:
|
||||
version: 0.16.0
|
||||
|
||||
- name: Install Rust
|
||||
uses: actions-rust-lang/setup-rust-toolchain@v1.15.4
|
||||
with:
|
||||
cache: true
|
||||
cache-on-failure: true
|
||||
cache-key: "v1-rust-stress-${{ matrix.os }}"
|
||||
components: rustfmt, clippy
|
||||
|
||||
- name: Stress test (seeded / deterministic)
|
||||
shell: bash
|
||||
run: make test-stress-seeded
|
||||
env:
|
||||
FFF_STRESS_CASES: "3"
|
||||
FFF_STRESS_MIN_OPS: "30"
|
||||
FFF_STRESS_MAX_OPS: "50"
|
||||
|
||||
- name: Stress test (random / fuzzy)
|
||||
shell: bash
|
||||
run: make test-stress-random
|
||||
env:
|
||||
FFF_STRESS_CASES: "5"
|
||||
FFF_STRESS_MIN_OPS: "30"
|
||||
FFF_STRESS_MAX_OPS: "60"
|
||||
|
||||
- name: Upload proptest regressions on failure
|
||||
if: failure()
|
||||
uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: proptest-regressions-${{ matrix.os }}
|
||||
path: crates/fff-core/tests/fuzz_git_watcher_stress.proptest-regressions
|
||||
if-no-files-found: ignore
|
||||
|
||||
fmt:
|
||||
name: cargo fmt
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@v5
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@master
|
||||
with:
|
||||
@@ -56,13 +118,13 @@ jobs:
|
||||
name: cargo clippy
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@v5
|
||||
|
||||
# Zig is required to compile zlob
|
||||
- name: Install Zig
|
||||
uses: mlugg/setup-zig@v2
|
||||
uses: goto-bus-stop/setup-zig@v2
|
||||
with:
|
||||
version: 0.15.2
|
||||
version: 0.16.0
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@master
|
||||
|
||||
@@ -17,7 +17,7 @@ jobs:
|
||||
name: Spell Check with Typos
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@v5
|
||||
with:
|
||||
persist-credentials: false
|
||||
|
||||
|
||||
@@ -25,7 +25,7 @@ jobs:
|
||||
name: Check lua files using Stylua
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@v5
|
||||
with:
|
||||
persist-credentials: false
|
||||
|
||||
|
||||
+15
-1
@@ -1,4 +1,5 @@
|
||||
doc/tags
|
||||
big-repo
|
||||
target/
|
||||
.archive.lua
|
||||
_*.lua
|
||||
@@ -10,6 +11,19 @@ result
|
||||
.repro/
|
||||
.wrangler/
|
||||
*.so
|
||||
big-repo/
|
||||
*.dylib
|
||||
# all the perf like utility files
|
||||
*.data
|
||||
node_modules/
|
||||
crates/fff-notify-debouncer-full/
|
||||
|
||||
dist/
|
||||
scripts/benchmark-results/
|
||||
|
||||
# Native binaries (downloaded at install)
|
||||
*.dylib
|
||||
*.so
|
||||
*.dll
|
||||
|
||||
# Instruments traces
|
||||
*.trace/
|
||||
|
||||
+25
@@ -0,0 +1,25 @@
|
||||
-- luacheck configuration for fff.nvim
|
||||
-- https://luacheck.readthedocs.io/en/stable/config.html
|
||||
|
||||
-- Neovim globals
|
||||
globals = { "vim" }
|
||||
|
||||
-- Standard library
|
||||
std = "luajit"
|
||||
|
||||
-- Ignore line length (handled by stylua)
|
||||
max_line_length = false
|
||||
|
||||
-- Ignore unused self argument in methods
|
||||
self = false
|
||||
|
||||
-- Files/directories to ignore
|
||||
exclude_files = {
|
||||
".luarocks/",
|
||||
}
|
||||
|
||||
-- Warn about unused variables, but allow _ prefix convention
|
||||
unused_args = true
|
||||
ignore = {
|
||||
"212", -- unused argument (too noisy for callback-heavy code)
|
||||
}
|
||||
@@ -0,0 +1,44 @@
|
||||
{
|
||||
"$schema": "https://raw.githubusercontent.com/LuaLS/vscode-lua/master/setting/schema.json",
|
||||
"runtime": {
|
||||
"version": "LuaJIT",
|
||||
"pathStrict": true
|
||||
},
|
||||
"workspace": {
|
||||
"library": [
|
||||
"/opt/nvim/share/nvim/runtime/lua/vim/_meta",
|
||||
"/opt/nvim/share/nvim/runtime/lua/vim/shared.lua",
|
||||
"${3rd}/luv/library",
|
||||
"${3rd}/busted/library",
|
||||
"/opt/snacks.nvim/lua"
|
||||
],
|
||||
"checkThirdParty": false
|
||||
},
|
||||
"diagnostics": {
|
||||
"severity": {
|
||||
"undefined-global": "Error",
|
||||
"undefined-field": "Warning",
|
||||
"missing-return": "Warning",
|
||||
"redundant-parameter": "Warning",
|
||||
"param-type-mismatch": "Warning",
|
||||
"assign-type-mismatch": "Warning",
|
||||
"cast-type-mismatch": "Warning",
|
||||
"deprecated": "Warning",
|
||||
"undefined-doc-param": "Warning"
|
||||
},
|
||||
"neededFileStatus": {
|
||||
"undefined-global": "Any",
|
||||
"undefined-field": "Any",
|
||||
"missing-return": "Any",
|
||||
"redundant-parameter": "Any",
|
||||
"param-type-mismatch": "Any",
|
||||
"assign-type-mismatch": "Any",
|
||||
"cast-type-mismatch": "Any",
|
||||
"deprecated": "Any",
|
||||
"undefined-doc-param": "Any"
|
||||
}
|
||||
},
|
||||
"type": {
|
||||
"checkTableShape": true
|
||||
}
|
||||
}
|
||||
+42
@@ -0,0 +1,42 @@
|
||||
{
|
||||
"$schema": "https://raw.githubusercontent.com/LuaLS/vscode-lua/master/setting/schema.json",
|
||||
"runtime": {
|
||||
"version": "LuaJIT"
|
||||
},
|
||||
"workspace": {
|
||||
"library": [
|
||||
"/opt/homebrew/share/nvim/runtime/lua/vim/_meta",
|
||||
"/opt/homebrew/share/nvim/runtime/lua/vim/shared.lua",
|
||||
"${3rd}/luv/library",
|
||||
"${3rd}/busted/library"
|
||||
],
|
||||
"checkThirdParty": false
|
||||
},
|
||||
"diagnostics": {
|
||||
"severity": {
|
||||
"undefined-global": "Error",
|
||||
"undefined-field": "Warning",
|
||||
"missing-return": "Warning",
|
||||
"redundant-parameter": "Warning",
|
||||
"param-type-mismatch": "Warning",
|
||||
"assign-type-mismatch": "Warning",
|
||||
"cast-type-mismatch": "Warning",
|
||||
"deprecated": "Warning",
|
||||
"undefined-doc-param": "Warning"
|
||||
},
|
||||
"neededFileStatus": {
|
||||
"undefined-global": "Any",
|
||||
"undefined-field": "Any",
|
||||
"missing-return": "Any",
|
||||
"redundant-parameter": "Any",
|
||||
"param-type-mismatch": "Any",
|
||||
"assign-type-mismatch": "Any",
|
||||
"cast-type-mismatch": "Any",
|
||||
"deprecated": "Any",
|
||||
"undefined-doc-param": "Any"
|
||||
}
|
||||
},
|
||||
"type": {
|
||||
"checkTableShape": true
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,9 @@
|
||||
{
|
||||
"mcpServers": {
|
||||
"fff": {
|
||||
"type": "stdio",
|
||||
"command": "./target/release/fff-mcp",
|
||||
"args": []
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,111 @@
|
||||
# To Clankers
|
||||
|
||||
This repository contains **FFF.nvim (Fast File Finder)**, a high-performance file picker for Neovim inspired by blink.cmp's fuzzy matching technology. It's NOT a completion plugin, but rather a standalone file finder with advanced fuzzy search and frecency scoring. The project aims to be the drop-in replacement for telescope, fzf-lua, snacks.picker and similar plugins, focusing on speed, accuracy search and usability features.
|
||||
|
||||
## Development Commands
|
||||
|
||||
Always prefer Makefile commands listed to the cargo/bun/node if possible.
|
||||
|
||||
### Building
|
||||
|
||||
- `make build` - build everything
|
||||
|
||||
### Testing and Development Tools
|
||||
|
||||
This project does not have a traditional test suite. Testing is done through:
|
||||
|
||||
- Create e2e local test file for Neovim: Load any Lua test file with `nvim -l <test_file>`
|
||||
- Write inline rust unit tests for any functionality that is standalone and scoped within a single function
|
||||
|
||||
### Code Quality
|
||||
|
||||
- `make lint` - Rust linting and code analysis
|
||||
- `make format` - Format all code
|
||||
- `make test` - Run unit tests (limited coverage, primarily integration testing)
|
||||
|
||||
When doing code make sure to REDUCE SIZE OF COMMENTS. This is very important. Every comment should be concise 1-2 liner maximum 4 lines if describes really extensive and unnatural concept.
|
||||
|
||||
### Important coding rules
|
||||
|
||||
- Do not add doc comments to the private structs and functions.
|
||||
- Do not make public structs if something can be private
|
||||
|
||||
|
||||
## Architecture
|
||||
|
||||
Everything that is performance critical happens in rust world, everything that is neovim specific happens in the lua code.
|
||||
|
||||
There are 3 main components:
|
||||
|
||||
- Rust binary with the global file picker state containing index of all files
|
||||
- Background thread with the file system watcher that updates the index in real time
|
||||
- Lua UI layer that renders the picker, handles user input, and calls the rust functions via FFI
|
||||
|
||||
There are 2 databases:
|
||||
|
||||
- Frecency database (LMDB) that tracks file access patterns for scoring
|
||||
- Query history database used to track the user's previous search queries
|
||||
|
||||
### Key Files
|
||||
|
||||
- `lua/fff.lua` - Entry point, delegates to main.lua
|
||||
- `lua/fff/main.lua` - Public API (find_files, search, change_directory)
|
||||
- `lua/fff/core.lua` - Initialization, autocmds, global state management
|
||||
- `lua/fff/picker_ui.lua` - UI rendering, layout calculation, keymaps
|
||||
- `lua/fff/file_picker/preview.lua` - File preview with syntax highlighting
|
||||
- `lua/fff/file_picker/image.lua` - Image preview (snacks.nvim integration)
|
||||
- `lua/fff/conf.lua` - Default config
|
||||
- `lua/fff/rust/init.lua` - Loads compiled Rust shared library
|
||||
|
||||
**Rust Side:**
|
||||
|
||||
- `lua/fff/rust/lib.rs` - FFI bindings, global state (FILE_PICKER, FRECENCY)
|
||||
- `lua/fff/rust/file_picker.rs` - Core FilePicker struct, indexing, background watcher
|
||||
- `lua/fff/rust/frecency.rs` - Frecency database (LMDB) and scoring
|
||||
- `lua/fff/rust/query_tracker.rs` - Search query history tracking
|
||||
- `lua/fff/rust/score.rs` - Fuzzy match scoring with frizbee integration
|
||||
- `lua/fff/rust/git.rs` - Git status caching and repository detection
|
||||
- `lua/fff/rust/background_watcher.rs` - File system watcher thread
|
||||
|
||||
### Scoring Algorithm
|
||||
|
||||
Located at the score.rs file
|
||||
|
||||
### Build System
|
||||
|
||||
- `Cargo.toml` - Rust dependencies and build configuration (package name: `fff_nvim`)
|
||||
- `rust-toolchain.toml` - Specifies Rust nightly toolchain with required components
|
||||
- `Cross.toml` - Cross-compilation settings using Zig for Linux targets
|
||||
- **CI/CD Workflows**:
|
||||
- `.github/workflows/rust.yml` - Rust testing, formatting, and clippy checks
|
||||
- `.github/workflows/release.yaml` - Automated multi-platform builds
|
||||
- `.github/workflows/stylua.yaml` - Lua code formatting validation
|
||||
- `.github/workflows/nix.yml` - Nix build validation
|
||||
- **Cross-compilation Support**: Uses `cross` tool with Zig backend for efficient cross-compilation
|
||||
|
||||
## Development Notes
|
||||
|
||||
### Working with Rust Code
|
||||
|
||||
- Prefer struct methods over functions
|
||||
- If there is more than 2 impls in the file - create new file
|
||||
- Smaller concise comments over giant comment blocks
|
||||
- Do not add doc comments to the private functions/structs
|
||||
- Be very careful around locking and better double check with the human if something is going to require potentially long lock on a mutex/rwlock
|
||||
|
||||
### Working with lua code
|
||||
|
||||
- Document the types of public functions in every module
|
||||
- Use `vim.validate()` for validating user inputs in public functions
|
||||
- Try to reuse as much of existing functions as possible
|
||||
- When working on new features for the UI **IT IS EXTREMELY IMPORTANT** to keep the core functionality of navigating between files, selecting, and seeing the preview working as is. NEVER break anything from the core UI functionality, only add new features on top of the current UI.
|
||||
- When making a large chunk of code make lua test that opens neovim at `~/dev/lightsource` and opens the picker to test the ui functionality across the actual code.
|
||||
- When adding a new highlights or any new shortcuts and configurable UI options add them to the neovim config. AND IMPORTANT: update the README.md with the new configuration options.
|
||||
|
||||
### UI rendering
|
||||
|
||||
When working on the UI changeds IT IS EXTREMELY important for you to test it for both prompt_position="bottom" and prompt_position="top" as the rendering logic is different for both of them in both rust and lua world. When the prompt is positioed in the bottom everything should work the same way as the top but would be reversed in order. (though navigation is same for both)
|
||||
|
||||
## Top level API that can not introduce breaking changes under any circumstance
|
||||
|
||||
Top level rust, lua, C, and bun APIs can not be changed under any circumstance
|
||||
Generated
+1260
-677
File diff suppressed because it is too large
Load Diff
+25
-8
@@ -2,16 +2,19 @@
|
||||
members = [
|
||||
"crates/fff-c",
|
||||
"crates/fff-core",
|
||||
"crates/fff-mcp",
|
||||
"crates/fff-nvim",
|
||||
"crates/fff-query-parser",
|
||||
"crates/fff-searcher",
|
||||
"crates/fff-grep",
|
||||
]
|
||||
resolver = "2"
|
||||
|
||||
[workspace.dependencies]
|
||||
fff-grep = { version = "0.8.1", path = "crates/fff-grep" }
|
||||
fff-query-parser = { version = "0.8.1", path = "crates/fff-query-parser", default-features = false }
|
||||
|
||||
# Shared dependencies
|
||||
ahash = "0.8"
|
||||
bindet = "0.3"
|
||||
blake3 = "1.8.2"
|
||||
chrono = { version = "0.4", features = ["serde"] }
|
||||
ctrlc = "3.4.2"
|
||||
@@ -22,23 +25,23 @@ git2 = { version = "0.20.2", default-features = false, features = [
|
||||
"vendored-libgit2",
|
||||
] }
|
||||
glidesort = "0.1"
|
||||
grep-matcher = "0.1.8"
|
||||
grep-searcher = { path = "crates/fff-searcher" }
|
||||
globset = "0.4"
|
||||
heed = "0.22.0"
|
||||
ignore = "0.4.22"
|
||||
memmap2 = "0.9"
|
||||
mimalloc = "0.1.47"
|
||||
zlob = "1.2.9"
|
||||
zlob = "1.3.3"
|
||||
|
||||
mlua = { version = "0.11.1", features = ["module", "luajit"] }
|
||||
neo_frizbee = "0.8.1"
|
||||
notify = "8.1.0"
|
||||
notify-debouncer-full = "0.7"
|
||||
neo_frizbee = { version = "0.10.2", features = ["match_end_col"] }
|
||||
notify = { version = "9.0.0-rc.3" }
|
||||
notify-debouncer-full = { package = "fff-notify-debouncer-full", version = "0.9.4" }
|
||||
once_cell = "1.20.2"
|
||||
parking_lot = "0.12"
|
||||
pathdiff = "0.2.1"
|
||||
rayon = "1.8.0"
|
||||
regex = "1.11"
|
||||
regex-syntax = "0.8"
|
||||
smallvec = { version = "1.13", features = ["const_generics", "union"] }
|
||||
thiserror = "2.0.10"
|
||||
tracing = "0.1"
|
||||
@@ -49,5 +52,19 @@ lto = "fat"
|
||||
codegen-units = 1
|
||||
strip = true
|
||||
|
||||
[profile.ci]
|
||||
inherits = "release"
|
||||
# we use lto=fat locally for better SIMD for the march=native but
|
||||
# on CI when we cross compiling we should not exclude any cpu flags checking
|
||||
lto = "thin"
|
||||
|
||||
[profile.bench]
|
||||
inherits = "release"
|
||||
|
||||
# For Instruments / xctrace: release-level optimization but keep debuginfo
|
||||
# and symbols so sampled frames resolve to real Rust names.
|
||||
[profile.prof]
|
||||
inherits = "release"
|
||||
debug = "full"
|
||||
strip = false
|
||||
lto = "thin"
|
||||
|
||||
@@ -1,28 +1,210 @@
|
||||
PLENARY_DIR ?= ../plenary.nvim
|
||||
MINI_DIR ?= ../mini.nvim
|
||||
|
||||
.PHONY: build test test-rust test-lua test-setup
|
||||
PREFIX ?= /usr/local
|
||||
LIBDIR ?= $(PREFIX)/lib
|
||||
INCLUDEDIR ?= $(PREFIX)/include
|
||||
|
||||
# Compile-time cfg that gates the watcher + git-status fuzz stress test.
|
||||
STRESS_RUSTFLAGS := --cfg stress
|
||||
FFF_STRESS_DEFAULT_SEED ?= 0xDEADBEEFCAFEBABE
|
||||
|
||||
SHELL := bash
|
||||
# Order matters: `-c` must be last so bash treats the recipe as the script
|
||||
# string rather than the literal `-o` / `pipefail` tokens.
|
||||
.SHELLFLAGS := -o pipefail -ec
|
||||
|
||||
.PHONY: build build-c-lib install uninstall test test-rust test-lua test-lua-snap test-version test-bun test-node prepare-bun prepare-node set-npm-version header test-stress test-stress-seeded test-stress-random test-stress-repos test-node-stress
|
||||
|
||||
all: format test lint
|
||||
|
||||
build:
|
||||
cargo build --release
|
||||
cargo build --release --features zlob
|
||||
|
||||
build-c-lib:
|
||||
cargo build --release -p fff-c --features zlob
|
||||
|
||||
header:
|
||||
cbindgen --config crates/fff-c/cbindgen.toml --crate fff-c --output crates/fff-c/include/fff.h
|
||||
|
||||
# Install the C library and header under $(PREFIX) (default /usr/local).
|
||||
# Override PREFIX for user-local installs, e.g. `make install PREFIX=$$HOME/.local`.
|
||||
# DESTDIR is honoured for packagers.
|
||||
install: build-c-lib
|
||||
install -d $(DESTDIR)$(LIBDIR)
|
||||
install -d $(DESTDIR)$(INCLUDEDIR)
|
||||
install -m 0644 crates/fff-c/include/fff.h $(DESTDIR)$(INCLUDEDIR)/fff.h
|
||||
@if [ -f target/release/libfff_c.dylib ]; then \
|
||||
install -m 0755 target/release/libfff_c.dylib $(DESTDIR)$(LIBDIR)/libfff_c.dylib; \
|
||||
echo "Installed $(DESTDIR)$(LIBDIR)/libfff_c.dylib"; \
|
||||
fi
|
||||
@if [ -f target/release/libfff_c.so ]; then \
|
||||
install -m 0755 target/release/libfff_c.so $(DESTDIR)$(LIBDIR)/libfff_c.so; \
|
||||
echo "Installed $(DESTDIR)$(LIBDIR)/libfff_c.so"; \
|
||||
fi
|
||||
@if [ -f target/release/fff_c.dll ]; then \
|
||||
install -m 0755 target/release/fff_c.dll $(DESTDIR)$(LIBDIR)/fff_c.dll; \
|
||||
echo "Installed $(DESTDIR)$(LIBDIR)/fff_c.dll"; \
|
||||
fi
|
||||
@echo "Installed header $(DESTDIR)$(INCLUDEDIR)/fff.h"
|
||||
|
||||
uninstall:
|
||||
rm -f $(DESTDIR)$(LIBDIR)/libfff_c.dylib
|
||||
rm -f $(DESTDIR)$(LIBDIR)/libfff_c.so
|
||||
rm -f $(DESTDIR)$(LIBDIR)/fff_c.dll
|
||||
rm -f $(DESTDIR)$(INCLUDEDIR)/fff.h
|
||||
@echo "Removed fff-c from $(DESTDIR)$(PREFIX)"
|
||||
|
||||
test-setup:
|
||||
@if [ ! -d "$(PLENARY_DIR)" ]; then \
|
||||
echo "Cloning plenary.nvim..."; \
|
||||
git clone --depth 1 https://github.com/nvim-lua/plenary.nvim $(PLENARY_DIR); \
|
||||
fi
|
||||
@if [ ! -d "$(MINI_DIR)" ]; then \
|
||||
echo "Cloning mini.nvim..."; \
|
||||
git clone --depth 1 https://github.com/echasnovski/mini.nvim $(MINI_DIR); \
|
||||
fi
|
||||
|
||||
test-rust:
|
||||
cargo test --verbose --workspace --exclude fff-nvim
|
||||
cargo test --workspace --features zlob --exclude fff-nvim
|
||||
|
||||
# neovim instance swallows internal crashes and doesn't rise the the error exiting silently
|
||||
# so check the stdout in case the sigsegv coming out of fff was printed (actual regression).
|
||||
# Output is streamed live via `tee`; pipefail (set above) propagates nvim's exit.
|
||||
test-lua: test-setup build
|
||||
@logfile=$$(mktemp); \
|
||||
trap 'rm -f "$$logfile"' EXIT; \
|
||||
nvim --headless -u tests/minimal_init.lua \
|
||||
-c "PlenaryBustedFile tests/fff_core_spec.lua" 2>&1
|
||||
-c "PlenaryBustedDirectory tests/ {minimal_init = 'tests/minimal_init.lua'}" 2>&1 \
|
||||
| tee "$$logfile"; \
|
||||
if grep -qE "SIG(SEGV|ABRT|BUS|FPE|ILL)" "$$logfile"; then \
|
||||
echo ""; \
|
||||
echo "FAIL: native crash detected during lua tests"; \
|
||||
exit 1; \
|
||||
fi
|
||||
|
||||
test: test-rust test-lua
|
||||
# mini.test reference_screenshot snapshots. Separate runner because mini.test
|
||||
# spawns child processes and uses its own collector (incompatible with
|
||||
# PlenaryBustedDirectory). Streams output live via `tee` so failure diffs
|
||||
# appear as they happen instead of after a long capture-buffered silence.
|
||||
# `pcall` catches collect-time errors (e.g. parse error in the test file)
|
||||
# that would otherwise leave headless nvim hanging in its event loop because
|
||||
# the reporter's `cquit` never fires.
|
||||
test-lua-snap: test-setup build
|
||||
@logfile=$$(mktemp); \
|
||||
trap 'rm -f "$$logfile"' EXIT; \
|
||||
nvim --headless -u tests/minimal_init.lua \
|
||||
-c "lua local ok,err=pcall(require('mini.test').run_file,'tests/picker_ui_snap.lua'); if not ok then io.stderr:write('mini.test failed to load: '..tostring(err)..'\\n'); vim.cmd('cquit 2') end" 2>&1 \
|
||||
| tee "$$logfile"; \
|
||||
if grep -qE "SIG(SEGV|ABRT|BUS|FPE|ILL)" "$$logfile"; then \
|
||||
echo ""; \
|
||||
echo "FAIL: native crash detected during snapshot tests"; \
|
||||
exit 1; \
|
||||
fi
|
||||
|
||||
test-version: test-setup
|
||||
nvim --headless -u tests/minimal_init.lua \
|
||||
-c "PlenaryBustedFile tests/version_spec.lua" 2>&1
|
||||
|
||||
prepare-bun: build
|
||||
mkdir -p packages/fff-bun/bin
|
||||
cp target/release/libfff_c.dylib packages/fff-bun/bin/ 2>/dev/null || true; \
|
||||
cp target/release/libfff_c.so packages/fff-bun/bin/ 2>/dev/null || true; \
|
||||
cp target/release/fff_c.dll packages/fff-bun/bin/ 2>/dev/null || true
|
||||
|
||||
prepare-node: build
|
||||
mkdir -p packages/fff-node/bin
|
||||
cp target/release/libfff_c.dylib packages/fff-node/bin/ 2>/dev/null || true; \
|
||||
cp target/release/libfff_c.so packages/fff-node/bin/ 2>/dev/null || true; \
|
||||
cp target/release/fff_c.dll packages/fff-node/bin/ 2>/dev/null || true
|
||||
|
||||
test-bun: prepare-bun
|
||||
cd packages/fff-bun && bun test src/
|
||||
cd packages/pi-fff && bun test test/
|
||||
|
||||
test-node: prepare-node
|
||||
cd packages/fff-node && npm run build && node test/e2e.mjs
|
||||
|
||||
# Bug pinning stress test script over fff-node for issue #515
|
||||
# Just keep it untouched because it's good enough + some stress for SDK
|
||||
FFF_STRESS_ITERS ?= 50
|
||||
test-node-stress: prepare-node
|
||||
cd packages/fff-node && npm run build && \
|
||||
FFF_STRESS_ITERS=$(FFF_STRESS_ITERS) node test/stress-515.mjs
|
||||
|
||||
test: test-rust test-lua test-lua-snap test-version test-bun test-node test-node-stress
|
||||
|
||||
test-stress-seeded:
|
||||
FFF_STRESS_SEED="$${FFF_STRESS_SEED:-$(FFF_STRESS_DEFAULT_SEED)}" \
|
||||
RUSTFLAGS="$(STRESS_RUSTFLAGS)" \
|
||||
cargo test --release \
|
||||
-p fff-search \
|
||||
--test fuzz_git_watcher_stress \
|
||||
--features zlob \
|
||||
-- --nocapture stress_seeded
|
||||
|
||||
test-stress-random:
|
||||
RUSTFLAGS="$(STRESS_RUSTFLAGS)" \
|
||||
cargo test --release \
|
||||
-p fff-search \
|
||||
--test fuzz_git_watcher_stress \
|
||||
--features zlob \
|
||||
-- --nocapture stress_random
|
||||
|
||||
test-stress-repos:
|
||||
RUSTFLAGS="$(STRESS_RUSTFLAGS)" \
|
||||
cargo test --release \
|
||||
-p fff-search \
|
||||
--test fuzz_real_repos \
|
||||
--features zlob \
|
||||
-- --nocapture
|
||||
|
||||
test-stress: test-stress-seeded test-stress-random test-stress-repos
|
||||
|
||||
# Update version in a package.json, including optionalDependencies.
|
||||
# Usage: make set-npm-version PKG=packages/fff-bun VERSION=1.0.0-nightly.abc1234
|
||||
set-npm-version:
|
||||
@test -n "$(PKG)" || (echo "PKG is required" && exit 1)
|
||||
@test -n "$(VERSION)" || (echo "VERSION is required" && exit 1)
|
||||
node -e " \
|
||||
const fs = require('fs'); \
|
||||
const pkg = JSON.parse(fs.readFileSync('$(PKG)/package.json', 'utf8')); \
|
||||
pkg.version = '$(VERSION)'; \
|
||||
if (pkg.optionalDependencies) { \
|
||||
for (const dep of Object.keys(pkg.optionalDependencies)) { \
|
||||
pkg.optionalDependencies[dep] = '$(VERSION)'; \
|
||||
} \
|
||||
} \
|
||||
fs.writeFileSync('$(PKG)/package.json', JSON.stringify(pkg, null, 2) + '\n'); \
|
||||
"
|
||||
@echo "Set $(PKG) to $(VERSION)"
|
||||
|
||||
format-rust:
|
||||
cargo fmt --all
|
||||
format-lua:
|
||||
stylua .
|
||||
format-ts:
|
||||
bun format
|
||||
|
||||
format: format-rust format-lua
|
||||
format: format-rust format-lua format-ts
|
||||
|
||||
lint-rust:
|
||||
cargo clippy --workspace --features zlob -- -D warnings
|
||||
lint-lua:
|
||||
~/.luarocks/bin/luacheck .
|
||||
lint-ts:
|
||||
bun lint
|
||||
|
||||
lint: lint-rust lint-lua lint-ts
|
||||
|
||||
check: format lint
|
||||
|
||||
CRATES_TO_PUBLISH= fff-grep fff-query-parser fff-search
|
||||
|
||||
publish-crates:
|
||||
@test -n "$(V)" || (echo "V is required. Usage: make publish-crates V=0.2.0" && exit 1)
|
||||
cargo install cargo-edit
|
||||
cargo set-version $(V) || exit 1;
|
||||
@for crate in $(CRATES_TO_PUBLISH); do \
|
||||
cargo publish -p $$crate --allow-dirty $$(if [ -n "$$CI" ]; then echo "--no-verify"; fi) || exit 1; \
|
||||
done
|
||||
|
||||
@@ -1,45 +1,113 @@
|
||||
<p align="center">
|
||||
<h2 align="center">FFF.nvim</h2>
|
||||
<img alt="FFF" src="./assets/logo-orange.png" width="300">
|
||||
|
||||
<p>
|
||||
<i>A file search toolkit for humans and AI agents. Really fast.</i>
|
||||
</p>
|
||||
|
||||
<p align="center">
|
||||
Finally a smart fuzzy file picker for neovim.
|
||||
</p>
|
||||
Typo-resistant path and content search, frecency-ranked file access, a background watcher, and a lightweight in-memory content index. Way faster than CLIs like ripgrep and fzf in any long-running process that searches more than once.
|
||||
|
||||
<p align="center" style="text-decoration: none; border: none;">
|
||||
<a href="https://github.com/dmtrKovalenko/fff.nvim/stargazers" style="text-decoration: none">
|
||||
<img alt="Stars" src="https://img.shields.io/github/stars/dmtrKovalenko/fff.nvim?style=for-the-badge&logo=starship&color=C9CBFF&logoColor=D9E0EE&labelColor=302D41"></a>
|
||||
<a href="https://github.com/dmtrKovalenko/fff.nvim/issues" style="text-decoration: none">
|
||||
<img alt="Issues" src="https://img.shields.io/github/issues/dmtrKovalenko/fff.nvim?style=for-the-badge&logo=bilibili&color=F5E0DC&logoColor=D9E0EE&labelColor=302D41"></a>
|
||||
<a href="https://github.com/dmtrKovalenko/fff.nvim/contributors" style="text-decoration: none"> <img alt="Contributors" src="https://img.shields.io/github/contributors/dmtrKovalenko/fff.nvim?color=%23DDB6F2&label=CONTRIBUTORS&logo=git&style=for-the-badge&logoColor=D9E0EE&labelColor=302D41"/></a>
|
||||
</p>
|
||||
Originally started as [Neovim plugin](#neovim-plugin) people loved, but it turned out that plenty of AI harnesses and code editors need the same thing: accurate, fast file search as a library. That is what fff is.
|
||||
|
||||
**FFF** stands for ~freakin fast fuzzy file finder~ (pick 3) and it is an opinionated fuzzy file picker for neovim. Just for files, but we'll try to solve file picking completely.
|
||||
---
|
||||
|
||||
It comes with a dedicated rust backend runtime that keep tracks of the file index, your file access and modifications, git status, and provides a comprehensive typo-resistant fuzzy search experience.
|
||||
Pick what you are interested in:
|
||||
|
||||
## Features
|
||||
<details id="mcp-server">
|
||||
<summary>
|
||||
<h2>MCP server</h2>
|
||||
</summary>
|
||||
|
||||
- Works out of the box with no additional configuration
|
||||
- [Typo resistant fuzzy search](https://github.com/saghen/frizbee)
|
||||
- Git status integration allowing to take advantage of last modified times within a worktree
|
||||
- Separate file index maintained by a dedicated backend allows <10 milliseconds search time for 50k files codebase
|
||||
- Display images in previews (for now requires snacks.nvim)
|
||||
- Smart in a plenty of different ways hopefully helpful for your workflow
|
||||
- This plugin initializes itself lazily by default
|
||||
Works with Claude Code, Codex, OpenCode, Cursor, Cline, and any MCP-capable client. Fewer grep roundtrips, less wasted context, faster answers.
|
||||
|
||||
## Installation
|
||||

|
||||
|
||||
> [!NOTE]
|
||||
> Although we'll try to make sure to keep 100% backward compatibility, by using you should understand that silly bugs and breaking changes may happen.
|
||||
> And also we hope for your contributions and feedback to make this plugin ideal for everyone.
|
||||
### One-line install
|
||||
|
||||
### Prerequisites
|
||||
Linux / macOS:
|
||||
|
||||
FFF.nvim requires:
|
||||
```bash
|
||||
curl -L https://dmtrkovalenko.dev/install-fff-mcp.sh | bash
|
||||
```
|
||||
|
||||
- Neovim 0.10.0+
|
||||
- [Rustup](https://rustup.rs/) (we require nightly for building the native backend rustup will handle toolchain automatically)
|
||||
Windows (PowerShell):
|
||||
|
||||
```powershell
|
||||
irm https://raw.githubusercontent.com/dmtrKovalenko/fff.nvim/main/install-mcp.ps1 | iex
|
||||
```
|
||||
|
||||
The scripts live at [`install-mcp.sh`](./install-mcp.sh) and [`install-mcp.ps1`](./install-mcp.ps1) if you want to read them first.
|
||||
|
||||
It prints the exact wiring instructions for your client. Once the server is connected, ask the agent to "use fff" and it picks up the `ffgrep`, `fffind`, and `fff-multi-grep` tools.
|
||||
|
||||
### Recommended agent prompt
|
||||
|
||||
Drop this into your project's `CLAUDE.md` or equivalent:
|
||||
|
||||
```markdown
|
||||
For any file search or grep in the current git-indexed directory, use fff tools.
|
||||
```
|
||||
|
||||
### What changes
|
||||
|
||||
- Frecency memory. Files you actually open rank higher next time. Warm-up from git touch history runs automatically.
|
||||
- Definition-first hinting. Lines that look like code definitions are classified on the Rust side, no regex overhead in your prompt.
|
||||
- Smart-case with auto-fuzzy fallback. `IsOffTheRecord` finds snake_case variants; zero-match queries retry as fuzzy and surface the best approximate hits.
|
||||
- Git-aware annotations. Modified, untracked, and staged files are tagged so the agent reaches for what you are actively changing.
|
||||
|
||||
Source: [`crates/fff-mcp/`](./crates/fff-mcp/).
|
||||
|
||||
</details>
|
||||
|
||||
The MCP server gives any agent a file search tool that is faster and more token-efficient than the built-in one.
|
||||
|
||||
<details id="pi-extension">
|
||||
<summary>
|
||||
<h2>Pi agent extension</h2>
|
||||
</summary>
|
||||
|
||||
### Install
|
||||
|
||||
```bash
|
||||
pi install npm:@ff-labs/pi-fff
|
||||
```
|
||||
|
||||
### Modes
|
||||
|
||||
Three operating modes, switchable at runtime with `/fff-mode`:
|
||||
|
||||
| Mode | What it does |
|
||||
| ------------------------ | --------------------------------------------------------------------------------- |
|
||||
| `tools-and-ui` (default) | Adds `ffgrep` and `fffind` tools, replaces `@`-mention autocomplete with FFF. |
|
||||
| `tools-only` | Only tool injection. Keeps pi's native editor autocomplete. |
|
||||
| `override` | Replaces pi's built-in `grep`, `find`, and `multi_grep` with FFF implementations. |
|
||||
|
||||
Env vars: `PI_FFF_MODE`, `FFF_FRECENCY_DB`, `FFF_HISTORY_DB`. Flags: `--fff-mode`, `--fff-frecency-db`, `--fff-history-db`.
|
||||
|
||||
### Agent-facing tools
|
||||
|
||||
- `ffgrep`. Content search. Accepts `path`, `exclude` (comma, space, or array; leading `!` optional), `caseSensitive`, `context`, and cursor pagination. Auto-detects regex, falls back to fuzzy on zero exact matches, rejects `.*`-style wildcard-only patterns up front.
|
||||
- `fffind`. Path and filename search. Matches the whole repo-relative path, not just the filename. Frecency-aware. The weak-match detector flags scattered fuzzy noise before it floods the agent's context.
|
||||
|
||||
### Commands
|
||||
|
||||
- `/fff-mode [tools-and-ui | tools-only | override]`. Show or switch the mode.
|
||||
- `/fff-health`. Picker, frecency, and git integration status.
|
||||
- `/fff-rescan`. Force a rescan.
|
||||
|
||||
Source: [`packages/pi-fff/`](./packages/pi-fff/).
|
||||
|
||||
</details>
|
||||
|
||||
The Pi extension swaps pi's native tools for FFF implementations and feeds the interactive editor's `@`-mention autocomplete from the frecency-ranked index.
|
||||
|
||||
<details id="neovim-plugin">
|
||||
<summary>
|
||||
<h2>fff.nvim</h2>
|
||||
</summary>
|
||||
|
||||
Demo on the Linux kernel repo (100k files, 8GB):
|
||||
|
||||
https://github.com/user-attachments/assets/5d0e1ce9-642c-4c44-aa88-01b05bb86abb
|
||||
|
||||
### Installation
|
||||
|
||||
@@ -49,42 +117,30 @@ FFF.nvim requires:
|
||||
{
|
||||
'dmtrKovalenko/fff.nvim',
|
||||
build = function()
|
||||
-- this will download prebuild binary or try to use existing rustup toolchain to build from source
|
||||
-- (if you are using lazy you can use gb for rebuilding a plugin if needed)
|
||||
-- downloads a prebuilt binary or falls back to cargo build
|
||||
require("fff.download").download_or_build_binary()
|
||||
end,
|
||||
-- if you are using nixos
|
||||
-- for nixos:
|
||||
-- build = "nix run .#release",
|
||||
opts = { -- (optional)
|
||||
opts = {
|
||||
debug = {
|
||||
enabled = true, -- we expect your collaboration at least during the beta
|
||||
show_scores = true, -- to help us optimize the scoring system, feel free to share your scores!
|
||||
enabled = true,
|
||||
show_scores = true,
|
||||
},
|
||||
},
|
||||
-- No need to lazy-load with lazy.nvim.
|
||||
-- This plugin initializes itself lazily.
|
||||
lazy = false,
|
||||
lazy = false, -- the plugin lazy-initialises itself
|
||||
keys = {
|
||||
{
|
||||
"ff", -- try it if you didn't it is a banger keybinding for a picker
|
||||
function() require('fff').find_files() end,
|
||||
desc = 'FFFind files',
|
||||
},
|
||||
{
|
||||
"fg",
|
||||
function() require('fff').live_grep() end,
|
||||
desc = 'LiFFFe grep',
|
||||
},
|
||||
{
|
||||
"fz",
|
||||
function() require('fff').live_grep({
|
||||
grep = {
|
||||
modes = { 'fuzzy', 'plain' }
|
||||
}
|
||||
}) end,
|
||||
{ "ff", function() require('fff').find_files() end, desc = 'FFFind files' },
|
||||
{ "fg", function() require('fff').live_grep() end, desc = 'LiFFFe grep' },
|
||||
{ "fz",
|
||||
function() require('fff').live_grep({ grep = { modes = { 'fuzzy', 'plain' } } }) end,
|
||||
desc = 'Live fffuzy grep',
|
||||
}
|
||||
}
|
||||
},
|
||||
{ "fc",
|
||||
function() require('fff').live_grep({ query = vim.fn.expand("<cword>") }) end,
|
||||
desc = 'Search current word',
|
||||
},
|
||||
},
|
||||
}
|
||||
```
|
||||
|
||||
@@ -94,386 +150,450 @@ FFF.nvim requires:
|
||||
vim.pack.add({ 'https://github.com/dmtrKovalenko/fff.nvim' })
|
||||
|
||||
vim.api.nvim_create_autocmd('PackChanged', {
|
||||
callback = function(event)
|
||||
if event.data.updated then
|
||||
callback = function(ev)
|
||||
local name, kind = ev.data.spec.name, ev.data.kind
|
||||
if name == 'fff.nvim' and (kind == 'install' or kind == 'update') then
|
||||
if not ev.data.active then vim.cmd.packadd('fff.nvim') end
|
||||
require('fff.download').download_or_build_binary()
|
||||
end
|
||||
end,
|
||||
})
|
||||
|
||||
-- the plugin will automatically lazy load
|
||||
vim.g.fff = {
|
||||
lazy_sync = true, -- start syncing only when the picker is open
|
||||
debug = {
|
||||
enabled = true,
|
||||
show_scores = true,
|
||||
},
|
||||
lazy_sync = true,
|
||||
debug = { enabled = true, show_scores = true },
|
||||
}
|
||||
|
||||
vim.keymap.set(
|
||||
'n',
|
||||
'ff',
|
||||
function() require('fff').find_files() end,
|
||||
{ desc = 'FFFind files' }
|
||||
)
|
||||
vim.keymap.set('n', 'ff', function() require('fff').find_files() end, { desc = 'FFFind files' })
|
||||
```
|
||||
|
||||
### Public API
|
||||
|
||||
```lua
|
||||
require('fff').find_files() -- find files in current repo
|
||||
require('fff').live_grep() -- live content grep
|
||||
require('fff').scan_files() -- force rescan
|
||||
require('fff').refresh_git_status() -- refresh git status
|
||||
require('fff').find_files_in_dir(path) -- find in a specific dir
|
||||
require('fff').change_indexing_directory(new_path) -- change root
|
||||
```
|
||||
|
||||
### Commands
|
||||
|
||||
- `:FFFScan`. Rescan files.
|
||||
- `:FFFRefreshGit`. Refresh git status.
|
||||
- `:FFFClearCache [all|frecency|files]`. Clear caches.
|
||||
- `:FFFHealth`. Health check.
|
||||
- `:FFFDebug [on|off|toggle]`. Toggle the scoring display.
|
||||
- `:FFFOpenLog`. Open `~/.local/state/nvim/log/fff.log`.
|
||||
|
||||
### Configuration
|
||||
|
||||
FFF.nvim comes with sensible defaults. Here's the complete configuration with all available options:
|
||||
Defaults are sensible. Override only what you care about.
|
||||
|
||||
```lua
|
||||
require('fff').setup({
|
||||
base_path = vim.fn.getcwd(),
|
||||
prompt = '🪿 ',
|
||||
title = 'FFFiles',
|
||||
max_results = 100,
|
||||
max_threads = 4,
|
||||
lazy_sync = true, -- set to false if you want file indexing to start on open
|
||||
layout = {
|
||||
height = 0.8,
|
||||
width = 0.8,
|
||||
prompt_position = 'bottom', -- or 'top'
|
||||
preview_position = 'right', -- or 'left', 'right', 'top', 'bottom'
|
||||
preview_size = 0.5,
|
||||
show_scrollbar = true, -- Show scrollbar for pagination
|
||||
-- How to shorten long directory paths in the file list:
|
||||
-- 'middle_number' (default): uses dots for 1-3 hidden (a/./b, a/../b, a/.../b)
|
||||
-- and numbers for 4+ (a/.4./b, a/.5./b)
|
||||
-- 'middle': always uses dots (a/./b, a/../b, a/.../b)
|
||||
-- 'end': truncates from the end (home/user/projects)
|
||||
path_shorten_strategy = 'middle_number',
|
||||
},
|
||||
preview = {
|
||||
enabled = true,
|
||||
max_size = 10 * 1024 * 1024, -- Do not try to read files larger than 10MB
|
||||
chunk_size = 8192, -- Bytes per chunk for dynamic loading (8kb - fits ~100-200 lines)
|
||||
binary_file_threshold = 1024, -- amount of bytes to scan for binary content (set 0 to disable)
|
||||
imagemagick_info_format_str = '%m: %wx%h, %[colorspace], %q-bit',
|
||||
line_numbers = false,
|
||||
wrap_lines = false,
|
||||
filetypes = {
|
||||
svg = { wrap_lines = true },
|
||||
markdown = { wrap_lines = true },
|
||||
text = { wrap_lines = true },
|
||||
},
|
||||
},
|
||||
keymaps = {
|
||||
close = '<Esc>',
|
||||
select = '<CR>',
|
||||
select_split = '<C-s>',
|
||||
select_vsplit = '<C-v>',
|
||||
select_tab = '<C-t>',
|
||||
-- you can assign multiple keys to any action
|
||||
move_up = { '<Up>', '<C-p>' },
|
||||
move_down = { '<Down>', '<C-n>' },
|
||||
preview_scroll_up = '<C-u>',
|
||||
preview_scroll_down = '<C-d>',
|
||||
toggle_debug = '<F2>',
|
||||
-- goes to the previous query in history
|
||||
cycle_previous_query = '<C-Up>',
|
||||
-- multi-select keymaps for quickfix
|
||||
toggle_select = '<Tab>',
|
||||
send_to_quickfix = '<C-q>',
|
||||
-- grep mode: cycle between plain text, regex, and fuzzy search
|
||||
toggle_grep_regex = '<S-Tab>',
|
||||
},
|
||||
hl = {
|
||||
border = 'FloatBorder',
|
||||
normal = 'Normal',
|
||||
cursor = 'CursorLine',
|
||||
matched = 'IncSearch',
|
||||
title = 'Title',
|
||||
prompt = 'Question',
|
||||
active_file = 'Visual',
|
||||
frecency = 'Number',
|
||||
debug = 'Comment',
|
||||
combo_header = 'Number',
|
||||
scrollbar = 'Comment', -- Highlight for scrollbar thumb (track uses border)
|
||||
directory_path = 'Comment', -- Highlight for directory path in file list
|
||||
-- Multi-select highlights
|
||||
selected = 'FFFSelected',
|
||||
selected_active = 'FFFSelectedActive',
|
||||
-- Git text highlights for file names
|
||||
git_staged = 'FFFGitStaged',
|
||||
git_modified = 'FFFGitModified',
|
||||
git_deleted = 'FFFGitDeleted',
|
||||
git_renamed = 'FFFGitRenamed',
|
||||
git_untracked = 'FFFGitUntracked',
|
||||
git_ignored = 'FFFGitIgnored',
|
||||
-- Git sign/border highlights
|
||||
git_sign_staged = 'FFFGitSignStaged',
|
||||
git_sign_modified = 'FFFGitSignModified',
|
||||
git_sign_deleted = 'FFFGitSignDeleted',
|
||||
git_sign_renamed = 'FFFGitSignRenamed',
|
||||
git_sign_untracked = 'FFFGitSignUntracked',
|
||||
git_sign_ignored = 'FFFGitSignIgnored',
|
||||
-- Git sign selected highlights
|
||||
git_sign_staged_selected = 'FFFGitSignStagedSelected',
|
||||
git_sign_modified_selected = 'FFFGitSignModifiedSelected',
|
||||
git_sign_deleted_selected = 'FFFGitSignDeletedSelected',
|
||||
git_sign_renamed_selected = 'FFFGitSignRenamedSelected',
|
||||
git_sign_untracked_selected = 'FFFGitSignUntrackedSelected',
|
||||
git_sign_ignored_selected = 'FFFGitSignIgnoredSelected',
|
||||
-- Grep highlights
|
||||
grep_match = 'IncSearch', -- Highlight for matched text in grep results
|
||||
grep_line_number = 'LineNr', -- Highlight for :line:col location
|
||||
grep_regex_active = 'DiagnosticInfo', -- Highlight for keybind + label when regex is on
|
||||
grep_regex_inactive = 'Comment', -- Highlight for keybind + label when regex is off
|
||||
-- Cross-mode suggestion highlights
|
||||
suggestion_header = 'WarningMsg', -- Highlight for the "No results found. Suggested..." banner
|
||||
},
|
||||
-- Store file open frecency
|
||||
frecency = {
|
||||
enabled = true,
|
||||
db_path = vim.fn.stdpath('cache') .. '/fff_nvim',
|
||||
},
|
||||
-- Store successfully opened queries with respective matches
|
||||
history = {
|
||||
enabled = true,
|
||||
db_path = vim.fn.stdpath('data') .. '/fff_queries',
|
||||
min_combo_count = 3, -- file will get a boost if it was selected 3 in a row times per specific query
|
||||
combo_boost_score_multiplier = 100, -- Score multiplier for combo matches
|
||||
},
|
||||
-- Git integration
|
||||
git = {
|
||||
status_text_color = false, -- Apply git status colors to filename text (default: false, only sign column)
|
||||
},
|
||||
debug = {
|
||||
enabled = false, -- Set to true to show scores in the UI
|
||||
show_scores = false,
|
||||
show_file_info = false, -- Show file info panel in preview
|
||||
},
|
||||
logging = {
|
||||
enabled = true,
|
||||
log_file = vim.fn.stdpath('log') .. '/fff.log',
|
||||
log_level = 'info',
|
||||
},
|
||||
-- Live grep search configuration
|
||||
grep = {
|
||||
max_file_size = 10 * 1024 * 1024, -- Skip files larger than 10MB
|
||||
max_matches_per_file = 200, -- Maximum matches per file
|
||||
smart_case = true, -- Case-insensitive unless query has uppercase
|
||||
time_budget_ms = 150, -- Max search time in ms per call (prevents UI freeze, 0 = no limit)
|
||||
modes = { 'plain', 'regex', 'fuzzy' }, -- Available grep modes and their cycling order
|
||||
}
|
||||
})
|
||||
```
|
||||
|
||||
### Key Features
|
||||
|
||||
#### Available Methods
|
||||
|
||||
```lua
|
||||
require('fff').find_files() -- Find files in current directory
|
||||
require('fff').find_in_git_root() -- Find files in the current git repository
|
||||
require('fff').scan_files() -- Trigger rescan of files in the current directory
|
||||
require('fff').refresh_git_status() -- Refresh git status for the active file lock
|
||||
require('fff').find_files_in_dir(path) -- Find files in a specific directory
|
||||
require('fff').change_indexing_directory(new_path) -- Change the base directory for the file picker
|
||||
```
|
||||
|
||||
#### Commands
|
||||
|
||||
FFF.nvim provides several commands for interacting with the file picker:
|
||||
|
||||
- `:FFFFind [path|query]` - Open file picker. Optional: provide directory path or search query
|
||||
- `:FFFScan` - Manually trigger a rescan of files in the current directory
|
||||
- `:FFFRefreshGit` - Manually refresh git status for all files
|
||||
- `:FFFClearCache [all|frecency|files]` - Clear various caches
|
||||
- `:FFFHealth` - Check FFF health status and dependencies
|
||||
- `:FFFDebug [on|off|toggle]` - Toggle debug scores display
|
||||
- `:FFFOpenLog` - Open the FFF log file in a new tab
|
||||
|
||||
#### Multiline Paste Support
|
||||
|
||||
The input field automatically handles multiline clipboard content by joining all lines into a single search query. This is particularly useful when copying file paths from terminal output.
|
||||
|
||||
#### Debug Mode
|
||||
|
||||
Toggle scoring information display:
|
||||
|
||||
- Press `F2` while in the picker
|
||||
- Use `:FFFDebug` command
|
||||
- Enable by default with `debug.show_scores = true`
|
||||
|
||||
#### Multi-Select and Quickfix Integration
|
||||
|
||||
Select multiple files and send them to Neovim's quickfix list (keymaps are configurable):
|
||||
|
||||
- `<Tab>` - Toggle selection for the current file (shows thick border `▊` in signcolumn)
|
||||
- `<C-q>` - Send selected files to quickfix list and close picker
|
||||
|
||||
#### Live Grep Search Modes
|
||||
|
||||
Live grep supports three search modes, cycled with `<S-Tab>`:
|
||||
|
||||
- **Plain text** (default) - The query is matched literally. Special regex characters like `.`, `*`, `(`, `)`, `$` have no special meaning. This is the safest mode for searching code containing regex metacharacters.
|
||||
- **Regex** - The query is interpreted as a regular expression. Supports character classes (`[a-z]`), quantifiers (`+`, `*`, `{n}`), alternation (`foo|bar`), anchors (`^`, `$`), word boundaries (`\b`), and more.
|
||||
- **Fuzzy** - The query is fuzzy matched using Smith-Waterman scoring. Accommodates typos and scattered characters (e.g., "mtxlk" matches "mutex_lock"). Results are filtered by a quality threshold to avoid overly fuzzy matches.
|
||||
|
||||
The current mode is shown on the right side of the input field (e.g., `plain`, `regex`, `fuzzy`) with color-coded highlighting.
|
||||
|
||||
You can customize which modes are available and their cycling order globally in your configuration, or per-call when invoking `live_grep()`.
|
||||
|
||||
**Global configuration:**
|
||||
|
||||
```lua
|
||||
require('fff').setup({
|
||||
grep = {
|
||||
modes = { 'plain', 'regex' }, -- Only plain and regex, no fuzzy
|
||||
}
|
||||
})
|
||||
```
|
||||
|
||||
**Per-call configuration:**
|
||||
|
||||
```lua
|
||||
-- Only fuzzy and plain modes for this specific grep
|
||||
require('fff').live_grep({
|
||||
grep = {
|
||||
modes = { 'fuzzy', 'plain' },
|
||||
}
|
||||
})
|
||||
|
||||
-- Single mode (hides mode indicator completely)
|
||||
require('fff').live_grep({
|
||||
grep = {
|
||||
modes = { 'fuzzy' },
|
||||
}
|
||||
})
|
||||
```
|
||||
|
||||
When only one mode is configured, the mode indicator is hidden completely and the cycle keybind does nothing.
|
||||
|
||||
#### Cross-Mode Suggestions
|
||||
|
||||
When a search returns no results, FFF automatically queries the opposite search mode and displays the results as suggestions:
|
||||
|
||||
- **File search with no matches** → shows suggested **content matches** (grep results) for the same query
|
||||
- **Grep search with no matches** → shows suggested **file name matches** for the same query
|
||||
|
||||
Suggestions are clearly labeled with a "No results found. Suggested ..." banner (highlighted with `hl.suggestion_header`). You can navigate and select suggestion items just like normal results — selecting a grep suggestion will open the file at the matching line.
|
||||
|
||||
#### Git Status Highlighting
|
||||
|
||||
FFF integrates with git to show file status through sign column indicators (enabled by default) and optional filename text coloring.
|
||||
|
||||
**Sign Column Indicators** (enabled by default) - Border characters shown in the sign column:
|
||||
|
||||
```lua
|
||||
hl = {
|
||||
git_sign_staged = 'FFFGitSignStaged',
|
||||
git_sign_modified = 'FFFGitSignModified',
|
||||
git_sign_deleted = 'FFFGitSignDeleted',
|
||||
git_sign_renamed = 'FFFGitSignRenamed',
|
||||
git_sign_untracked = 'FFFGitSignUntracked',
|
||||
git_sign_ignored = 'FFFGitSignIgnored',
|
||||
}
|
||||
```
|
||||
|
||||
**Text Highlights** (opt-in) - Apply colors to filenames based on git status:
|
||||
|
||||
To enable git status text coloring, set `git.status_text_color = true`:
|
||||
|
||||
```lua
|
||||
require('fff').setup({
|
||||
git = {
|
||||
status_text_color = true, -- Enable git status colors on filename text
|
||||
base_path = vim.fn.getcwd(),
|
||||
prompt = '> ',
|
||||
title = 'FFFiles',
|
||||
max_results = 100,
|
||||
max_threads = 4,
|
||||
lazy_sync = true,
|
||||
prompt_vim_mode = false,
|
||||
layout = {
|
||||
height = 0.8,
|
||||
width = 0.8,
|
||||
prompt_position = 'bottom', -- or 'top'
|
||||
preview_position = 'right', -- 'left' | 'right' | 'top' | 'bottom'
|
||||
preview_size = 0.5,
|
||||
flex = { size = 130, wrap = 'top' },
|
||||
min_list_height = 10, -- do not display anything except the list below this threshold
|
||||
show_scrollbar = true,
|
||||
path_shorten_strategy = 'middle_number', -- 'middle_number' | 'middle' | 'end' | 'start'
|
||||
anchor = 'center',
|
||||
},
|
||||
preview = {
|
||||
enabled = true,
|
||||
max_size = 10 * 1024 * 1024,
|
||||
chunk_size = 8192,
|
||||
binary_file_threshold = 1024,
|
||||
imagemagick_info_format_str = '%m: %wx%h, %[colorspace], %q-bit',
|
||||
line_numbers = false,
|
||||
cursorlineopt = 'both',
|
||||
wrap_lines = false,
|
||||
filetypes = {
|
||||
svg = { wrap_lines = true },
|
||||
markdown = { wrap_lines = true },
|
||||
text = { wrap_lines = true },
|
||||
},
|
||||
},
|
||||
keymaps = {
|
||||
close = '<Esc>',
|
||||
select = '<CR>',
|
||||
select_split = '<C-s>',
|
||||
select_vsplit = '<C-v>',
|
||||
select_tab = '<C-t>',
|
||||
move_up = { '<Up>', '<C-p>' },
|
||||
move_down = { '<Down>', '<C-n>' },
|
||||
preview_scroll_up = '<C-u>',
|
||||
preview_scroll_down = '<C-d>',
|
||||
toggle_debug = '<F2>',
|
||||
cycle_grep_modes = '<S-Tab>',
|
||||
cycle_previous_query = '<C-Up>',
|
||||
toggle_select = '<Tab>',
|
||||
send_to_quickfix = '<C-q>',
|
||||
focus_list = '<leader>l',
|
||||
focus_preview = '<leader>p',
|
||||
},
|
||||
frecency = {
|
||||
enabled = true,
|
||||
db_path = vim.fn.stdpath('cache') .. '/fff_nvim',
|
||||
},
|
||||
history = {
|
||||
enabled = true,
|
||||
db_path = vim.fn.stdpath('data') .. '/fff_queries',
|
||||
min_combo_count = 3,
|
||||
combo_boost_score_multiplier = 100,
|
||||
},
|
||||
git = {
|
||||
status_text_color = false, -- true to color filenames by git status
|
||||
},
|
||||
grep = {
|
||||
max_file_size = 10 * 1024 * 1024,
|
||||
max_matches_per_file = 100,
|
||||
smart_case = true,
|
||||
time_budget_ms = 150,
|
||||
modes = { 'plain', 'regex', 'fuzzy' },
|
||||
trim_whitespace = false,
|
||||
},
|
||||
debug = {
|
||||
enabled = false, -- show the file info panel next to the preview
|
||||
show_scores = false, -- inline scores in the file list
|
||||
-- Per-section toggles for the file info panel. Accepts a boolean shorthand
|
||||
-- (`show_file_info = true|false`) to flip everything at once. The panel
|
||||
-- adapts to width: narrow renders sections vertically, wide renders them
|
||||
-- as a two-column grid. Disable a section to also shrink the panel.
|
||||
show_file_info = {
|
||||
file_info = true, -- size, type, git status, frecency
|
||||
score_breakdown = true, -- total + match type, bonuses, modifiers, penalty
|
||||
-- modified + accessed timestamps; pass a table to hide individual rows:
|
||||
-- timings = { modified = false, accessed = true }
|
||||
timings = true,
|
||||
full_path = true, -- relative path at the bottom (wraps if too long)
|
||||
},
|
||||
},
|
||||
logging = {
|
||||
enabled = true,
|
||||
log_file = vim.fn.stdpath('log') .. '/fff.log',
|
||||
log_level = 'info',
|
||||
},
|
||||
hl = {
|
||||
git_staged = 'FFFGitStaged', -- Files staged for commit
|
||||
git_modified = 'FFFGitModified', -- Modified unstaged files
|
||||
git_deleted = 'FFFGitDeleted', -- Deleted files
|
||||
git_renamed = 'FFFGitRenamed', -- Renamed files
|
||||
git_untracked = 'FFFGitUntracked', -- New untracked files
|
||||
git_ignored = 'FFFGitIgnored', -- Git-ignored files
|
||||
}
|
||||
})
|
||||
```
|
||||
|
||||
The plugin provides sensible default highlight groups that link to common git highlight groups (e.g., GitSignsAdd, GitSignsChange). You can override these with your own custom highlight groups to match your colorscheme.
|
||||
### Live grep modes
|
||||
|
||||
**Example - Custom Bright Colors for Text:**
|
||||
`<S-Tab>` cycles between `plain`, `regex`, and `fuzzy`. The list is configurable via `grep.modes`, and single-mode setups hide the indicator entirely.
|
||||
|
||||
Per-call override:
|
||||
|
||||
```lua
|
||||
vim.api.nvim_set_hl(0, 'CustomGitModified', { fg = '#FFA500' })
|
||||
vim.api.nvim_set_hl(0, 'CustomGitUntracked', { fg = '#00FF00' })
|
||||
|
||||
require('fff').setup({
|
||||
git = {
|
||||
status_text_color = true,
|
||||
},
|
||||
hl = {
|
||||
git_modified = 'CustomGitModified',
|
||||
git_untracked = 'CustomGitUntracked',
|
||||
}
|
||||
})
|
||||
require('fff').live_grep({ grep = { modes = { 'fuzzy', 'plain' } } })
|
||||
require('fff').live_grep({ query = 'search term' }) -- pre-fill
|
||||
```
|
||||
|
||||
#### File Filtering
|
||||
### Constraints
|
||||
|
||||
FFF.nvim respects `.gitignore` patterns automatically. To filter files from the picker without modifying `.gitignore`, create a `.ignore` file in your project root:
|
||||
Both find and grep accept these tokens to refine a query:
|
||||
|
||||
- `git:modified`. One of `modified`, `staged`, `deleted`, `renamed`, `untracked`, `ignored`.
|
||||
- `test/`. Any deeply nested children of `test/`.
|
||||
- `!something`, `!test/`, `!git:modified`. Exclusion.
|
||||
- `./**/*.{rs,lua}`. Any valid glob, powered by [zlob](https://github.com/dmtrKovalenko/zlob).
|
||||
|
||||
Grep-only:
|
||||
|
||||
- `*.md`, `*.{c,h}`. Extension filter.
|
||||
- `src/main.rs`. Grep inside a single file.
|
||||
|
||||
Mix freely: `git:modified src/**/*.rs !src/**/mod.rs user controller`.
|
||||
|
||||
### Multi-select and quickfix
|
||||
|
||||
- `<Tab>`. Toggle selection (shows a thick `▊` in the signcolumn).
|
||||
- `<C-q>`. Send selected files to the quickfix list and close the picker.
|
||||
|
||||
### Git status highlighting
|
||||
|
||||
Sign-column indicators are on by default. To color filename text by git status, set `git.status_text_color = true` and adjust the `hl.git_*` groups. See `:help fff.nvim` for the full list.
|
||||
|
||||
### Float colors
|
||||
|
||||
The picker maps its float content to `NormalFloat` (via `hl.normal`) and the border to `FloatBorder`. Default `FloatBorder` links to `NormalFloat`, so border and content share a background out of the box and the picker reads as a single popup. Override `hl.normal = 'Normal'` to make the picker blend with the editor instead.
|
||||
|
||||
### File info panel
|
||||
|
||||
Enable with `debug.enabled = true`. The panel sits above the preview and shows
|
||||
file metadata, score breakdown, timestamps and the full absolute path. It
|
||||
adapts to the panel width: at narrow widths sections stack vertically (B2),
|
||||
at wide widths sections render as a two-column grid (H2). Each section can be
|
||||
disabled individually via `debug.show_file_info`.
|
||||
|
||||
Customise the panel via `hl`:
|
||||
|
||||
| key | default | used for |
|
||||
| ---------------------------- | -------------------- | ----------------------------------- |
|
||||
| `file_info_section` | `Title` | section header label |
|
||||
| `file_info_separator` | `FloatBorder` | dashes that act as section borders |
|
||||
| `file_info_label` | `Comment` | row labels (Size, Type, Git, ...) |
|
||||
| `file_info_value` | `Normal` fg | plain values |
|
||||
| `file_info_value_dim` | `NonText` | dim values, separators inside rows |
|
||||
| `file_info_size` | `Number` | file size value |
|
||||
| `file_info_type` | `Type` | filetype value |
|
||||
| `file_info_path` | `Directory` | full path |
|
||||
| `file_info_total_score` | bold + `Number` | total score (bold) |
|
||||
| `file_info_match_type` | bold + `Special` | match type (bold) |
|
||||
| `file_info_score_pos` | `DiagnosticOk` | positive score components |
|
||||
| `file_info_score_neg` | `DiagnosticError` | negative score components |
|
||||
|
||||
### File filtering
|
||||
|
||||
FFF honours `.gitignore`. For picker-only ignores that do not touch git, add a sibling `.ignore` file:
|
||||
|
||||
```gitignore
|
||||
# Exclude all markdown files
|
||||
*.md
|
||||
|
||||
# Exclude specific subdirectory
|
||||
docs/archive/**/*.md
|
||||
```
|
||||
|
||||
Run `:FFFScan` to force a rescan if needed.
|
||||
Run `:FFFScan` to force a rescan.
|
||||
|
||||
### Troubleshooting
|
||||
|
||||
#### Health Check
|
||||
- `:FFFHealth` verifies picker init, optional dependencies, and DB connectivity.
|
||||
- `:FFFOpenLog` opens the log file.
|
||||
|
||||
Run `:FFFHealth` to check the status of FFF.nvim and its dependencies. This will verify:
|
||||
</details>
|
||||
|
||||
- File picker initialization status
|
||||
- Optional dependencies (git, image preview tools)
|
||||
- Database connectivity
|
||||
The best file search picker for neovim. Period. Faster and more intuitive queries, frecency ranking, definition classification and much more.
|
||||
|
||||
#### Viewing Logs
|
||||
<details id="node-sdk">
|
||||
<summary>
|
||||
<h2>Node & Bun SDK</h2>
|
||||
</summary>
|
||||
|
||||
If you encounter issues, check the log file:
|
||||
|
||||
```vim
|
||||
:FFFOpenLog
|
||||
```bash
|
||||
npm install @ff-labs/fff-node
|
||||
# or
|
||||
bun add @ff-labs/fff-node
|
||||
```
|
||||
|
||||
Or manually open the log file at `~/.local/state/nvim/log/fff.log` (default location).
|
||||
```ts
|
||||
import { FileFinder } from "@ff-labs/fff-node";
|
||||
|
||||
#### Common Issues
|
||||
const finder = FileFinder.create({ basePath: process.cwd(), aiMode: true });
|
||||
if (!finder.ok) throw new Error(finder.error);
|
||||
await finder.value.waitForScan(10_000);
|
||||
|
||||
**File picker not initializing:**
|
||||
const files = finder.value.fileSearch("incognito profile", { pageSize: 20 });
|
||||
const hits = finder.value.grep("GetOffTheRecordProfile", {
|
||||
mode: "plain",
|
||||
smartCase: true,
|
||||
beforeContext: 1,
|
||||
afterContext: 1,
|
||||
classifyDefinitions: true,
|
||||
});
|
||||
|
||||
- Ensure the Rust backend is compiled: `cargo build --release` in the plugin directory
|
||||
- Check that your Neovim version is 0.10.0 or higher
|
||||
finder.value.destroy();
|
||||
```
|
||||
|
||||
**Image previews not working:**
|
||||
Every method returns a `Result<T>` (`{ ok: true, value } | { ok: false, error }`). Full type reference: [`packages/fff-node/src/types.ts`](./packages/fff-node/src/types.ts).
|
||||
|
||||
- Verify your terminal supports images (kitty, iTerm2, WezTerm, etc.)
|
||||
- For terminals without native image support, install one of: `chafa`, `viu`, or `img2txt`
|
||||
- If using snacks.nvim, ensure it's properly configured
|
||||
</details>
|
||||
|
||||
**Performance issues:**
|
||||
TypeScript wrapper over the C library for nodejs and bun. Build custom agent tools, CLIs, or IDE integrations on top of FFF.
|
||||
|
||||
- Adjust `max_threads` in configuration based on your system
|
||||
- Reduce `preview.max_lines` and `preview.max_size` for large files
|
||||
- Clear cache if it becomes too large: `:FFFClearCache all`
|
||||
<details id="rust-crate">
|
||||
<summary>
|
||||
<h2>Rust crate</h2>
|
||||
</summary>
|
||||
|
||||
**Files not being indexed:**
|
||||
### Add the dependency
|
||||
|
||||
- Run `:FFFScan` to manually trigger a file scan
|
||||
- Check that the `base_path` is correctly set
|
||||
- Verify you have read permissions for the directory
|
||||
FFF is written in Rust, so this is the lowest-overhead way to use it.
|
||||
|
||||
#### Debug Mode
|
||||
```toml
|
||||
[dependencies]
|
||||
fff-search = "0.6"
|
||||
```
|
||||
|
||||
Enable debug mode to see scoring information and troubleshoot search results:
|
||||
Full API documentation: [docs.rs/fff-search](https://docs.rs/fff-search/latest/fff_search/).
|
||||
|
||||
- Press `F2` while in the picker
|
||||
- Run `:FFFDebug on` to enable permanently
|
||||
- Set `debug.show_scores = true` in configuration
|
||||
</details>
|
||||
|
||||
Native rust crate that is performing all the search. Stable and well documented.
|
||||
|
||||
<details id="c-library">
|
||||
<summary>
|
||||
<h2>C library</h2>
|
||||
</summary>
|
||||
|
||||
### Build
|
||||
|
||||
```bash
|
||||
# Builds only the C cdylib (fastest):
|
||||
make build-c-lib
|
||||
|
||||
# or directly with cargo:
|
||||
cargo build --release -p fff-c --features zlob
|
||||
```
|
||||
|
||||
The output is a `cdylib` (`libfff_c.so` / `libfff_c.dylib` / `fff_c.dll`). The header lives at [`crates/fff-c/include/fff.h`](./crates/fff-c/include/fff.h).
|
||||
|
||||
Prebuilt binaries for every version, including every commit on main, are on the [releases page](https://github.com/dmtrKovalenko/fff.nvim/releases). The same binaries also ship inside the `@ff-labs/fff-bin-*` npm packages.
|
||||
|
||||
### Install
|
||||
|
||||
```bash
|
||||
# System-wide (needs sudo):
|
||||
sudo make install
|
||||
|
||||
# User-local, no sudo:
|
||||
make install PREFIX=$HOME/.local
|
||||
|
||||
# Staged install for packagers:
|
||||
make install DESTDIR=/tmp/pkgroot PREFIX=/usr
|
||||
```
|
||||
|
||||
Drops `libfff_c.{so,dylib,dll}` into `$(PREFIX)/lib` and the header into `$(PREFIX)/include/fff.h`. Remove with `make uninstall`, which honours the same `PREFIX` and `DESTDIR`.
|
||||
|
||||
Link against it after install:
|
||||
|
||||
```bash
|
||||
cc my_app.c -lfff_c -o my_app
|
||||
```
|
||||
|
||||
Ensure `$(PREFIX)/lib` is on your runtime library search path (`LD_LIBRARY_PATH` on Linux, `DYLD_LIBRARY_PATH` on macOS, or an entry in `/etc/ld.so.conf.d/`).
|
||||
|
||||
### Minimal example
|
||||
|
||||
```c
|
||||
#include <fff.h>
|
||||
#include <stdio.h>
|
||||
|
||||
int main(void) {
|
||||
FffResult *res = fff_create_instance(
|
||||
".", // base_path
|
||||
"", // frecency_db_path (empty = default)
|
||||
"", // history_db_path
|
||||
false, // use_unsafe_no_lock
|
||||
true, // enable_mmap_cache
|
||||
true, // enable_content_indexing
|
||||
true, // watch
|
||||
false // ai_mode
|
||||
);
|
||||
if (!res->success) {
|
||||
fprintf(stderr, "init failed: %s\n", res->error);
|
||||
fff_free_result(res);
|
||||
return 1;
|
||||
}
|
||||
void *handle = res->handle;
|
||||
fff_free_result(res);
|
||||
|
||||
// Search
|
||||
FffResult *search = fff_search(handle, "main.rs", "", 0, 0, 20, 100, 3);
|
||||
// ... read FffSearchResult from search->handle, then fff_free_search_result()
|
||||
|
||||
fff_destroy(handle);
|
||||
return 0;
|
||||
}
|
||||
```
|
||||
|
||||
### Notes
|
||||
|
||||
- Every function returning `FffResult*` allocates with Rust's `Box`. Free with `fff_free_result`, do not use malloc's free
|
||||
- Payloads (search results, grep results, scan progress) have their own dedicated free functions listed in the header.
|
||||
- C strings returned in the `handle` field (e.g. from `fff_get_base_path`) are freed with `fff_free_string`.
|
||||
|
||||
Source: [`crates/fff-c/`](./crates/fff-c/).
|
||||
|
||||
</details>
|
||||
|
||||
Stable C ABI. Bind from C/C++, Zig, Go via cgo, Python via ctypes, or anything with C FFI.
|
||||
|
||||
---
|
||||
|
||||
## What is FFF and why use it over ripgrep or fzf?
|
||||
|
||||
FFF is a file search library, not a CLI. Ripgrep and fzf are great tools, but they are command-line programs: every call forks a new process, re-reads `.gitignore`, re-stats directories, and rebuilds whatever state it needs in memory before it can answer. That is fine when you grep once from a shell. It is bad when an editor or an AI agent wants to run hundreds of searches per session.
|
||||
|
||||
FFF keeps the index and the file cache resident in one long-lived process and exposes the same Rust core through four thin layers: a native crate (`fff-search`), a C library (`libfff_c`), a Node/Bun SDK (`@ff-labs/fff-node`), and an MCP server. You call `FileFinder.create()` once, then every subsequent search hits warm memory. On a 500k-file Chromium checkout, that is the difference between 3-9 **SECONDS** per ripgrep spawn and sub-10 ms per FFF query.
|
||||
|
||||
Algorithm for fuzzy matching is much more comprehensive than fzf's algorithm it is **typo-resistant** and we provide a query language with additional constraint parsing for prefiltering e.g. "*.rs !test/ shcema" is a perfectly valid query for fff, but fzf wouldn't find anything even for a single typo in "shcema".
|
||||
|
||||
### Why a programmatic API matters
|
||||
|
||||
- No process spawn. Every call stays in-process and avoids the fork, exec, argv parsing, and stdout pipe setup that dominates short `rg` invocations.
|
||||
- One FS walk, metadata collection, and parse of `.gitignore`. The ignore walker runs once at scan time and the result is reused for every search.
|
||||
- Results come back as typed objects, not text you have to re-parse. The SDK gives you `{ relativePath, lineNumber, lineContent, gitStatus, totalFrecencyScore, isDefinition, ... }` directly.
|
||||
- Cursor pagination that survives across calls. Ripgrep has no concept of "page 2 of these matches"; FFF does.
|
||||
- A long-lived process opens up optimisations that a one-shot CLI cannot apply: warm caches, incremental re-indexing, cross-query frecency, and shared SIMD state.
|
||||
|
||||
### What the core actually does
|
||||
|
||||
- **Frecency-ranked fuzzy matching.** Every indexed file carries an access score and a modification score. Searches rank files you have opened recently and frequently above cold results. This is the same idea as VS Code's recently-opened list, but applied to every search result, not just a sidebar.
|
||||
- **Typo-resistant matching for both paths and content.** Smith-Waterman fuzzy scoring is available on the grep path; path search uses SIMD-accelerated fuzzy matching (via the [`frizbee`](https://github.com/saghm/frizbee)-derived core) that survives dropped characters and reorderings.
|
||||
- **Content grep with three modes.** Plain literal (SIMD memmem), regex (the Rust `regex` crate), and fuzzy (Smith-Waterman per line). Auto-detects which mode to use from the pattern, falls back to fuzzy when a plain search returns zero hits.
|
||||
- **Multi-pattern OR search.** SIMD Aho-Corasick for "find any of these 20 identifiers at once", which is faster than regex alternation and a lot faster than 20 separate ripgrep runs.
|
||||
- **Background file watcher.** The index updates as files change. You never pay for a rescan on the hot path.
|
||||
- **Git status awareness.** Modified, staged, untracked, and ignored states are cached and returned with every result, so callers can sort or filter them without shelling out to git. The watcher talks to libgit2 directly instead of spawning the `git` CLI.
|
||||
- **Definition classifier.** A byte-level scanner on the Rust side tags lines that start with `struct`, `fn`, `class`, `def`, `impl`, and friends.
|
||||
|
||||
### Performance choices that matter
|
||||
|
||||
- Efficient memory allocator and memory allocation strategy (see next paragraph). By default we use `mimaloc`
|
||||
- Parallel multi thread search pipeline that is not contaganted by the orchistration logic
|
||||
- SIMD first algorithms for everything. Efficinet & non-allocating sorting.
|
||||
- Platform specific optimizations for FS ([getdents64](https://linux.die.net/man/2/getdents64), NTFS api on windows and others)
|
||||
- Lightweight on the flight content index for realtime even typo resistant grep
|
||||
- Memory mapped content cache. We store some of the files in virtual memory (the amount is limited)
|
||||
- Single contiguous arena storage of string chunks. Significantly reduces the amount of memory to work with and dramatically increases CPU cache hits.
|
||||
|
||||
### Memory allocation
|
||||
|
||||
Yes, fff fundamentally requires more memory than calling a single child process. That is the primary source of the speedup. In practice, alongside one of the most popular file search pickers for Neovim, [fff ends up using less RAM than a burst of ripgrep invocations](https://x.com/neogoose_btw/status/2041606853155811442).
|
||||
|
||||
|
||||
FFF also keeps a content index, around 360 bytes per indexed file, so roughly 36 MB for a 100k-file repo. Not every file is indexed - binaries, oversized files, and anything not eligible for grep are skipped. If even that footprint is too much, the index can be backed by a memory-mapped file instead of anonymous RAM.
|
||||
|
||||
### What this means in practice
|
||||
|
||||
If you are building an agent, an IDE extension, a pre-commit check, or any long-running tool that searches the same repository many times, calling FFF as a library is dramatically cheaper than shelling out to ripgrep. The tradeoff is real memory: FFF keeps the index in RAM and warms the content cache. On a 14k-file repo that costs about 26 MB resident. On a 500k-file repo like Chromium, expect a few hundred MB. In exchange, every single search is enriched with git status, frecency ranking, file metadata, timestamps of last access and edit and so on.
|
||||
|
||||
If you are running one grep from a terminal, `rg` is still the right tool. If you run dozens of them inside the same process, FFF will pay for itself starting from the second call. If you work on AI agent fff will finish preparation work before your AI will have a chance to call it.
|
||||
|
||||
### How it compares
|
||||
|
||||
- **ripgrep**: FFF uses the same underlying regex engine and more advanced plain text matching algorithms. Stores content index and file tree. Main wins on repeated-search workloads. Loses on "grep once from bash and exit."
|
||||
- **fzf**: FFF's path search is fuzzy like fzf, but it is also frecency-aware and git-aware, and ships a more typo-tolerant algorithm. fzf is a pure match-and-filter tool; FFF ranks results by how often you actually open them.
|
||||
- **Telescope / fzf-lua / snacks.picker**: FFF ships its own Neovim picker with the same ranking the MCP server and SDK use. The picker is optional; the core is the same.
|
||||
- **Tantivy or other full-text search engines**: different class of tool. Tantivy indexes documents for query-time scoring at scale. FFF is scoped to one repository and optimised for sub-10 ms response. It does not persist an inverted index on disk.
|
||||
|
||||
---
|
||||
|
||||
## Repository layout
|
||||
|
||||
- `crates/fff-search`, `crates/fff-grep`, `crates/fff-query-parser` - Rust core.
|
||||
- `crates/fff-c` - C FFI used by every language binding.
|
||||
- `crates/fff-nvim` - Lua/mlua bindings for the Neovim plugin.
|
||||
- `crates/fff-mcp` - MCP server binary.
|
||||
- `packages/fff-node` - Node.js SDK (`@ff-labs/fff-node`).
|
||||
- `packages/fff-bun` - Bun SDK (`@ff-labs/fff-node`).
|
||||
- `packages/pi-fff` - pi extension (`@ff-labs/pi-fff`).
|
||||
- `lua/` - Neovim-side plugin code.
|
||||
|
||||
## Contributing
|
||||
|
||||
Bug reports and pull requests welcome. Agentic coding tools are welcome to be used, but human review is mandatory.
|
||||
|
||||
## License
|
||||
|
||||
[MIT](./LICENSE) & open source forever.
|
||||
|
||||
@@ -4,6 +4,14 @@ extend-exclude = ["/CHANGELOG.md", "data/filetypes/base.lua"]
|
||||
[default.extend-words]
|
||||
noice = "noice"
|
||||
fo = "fo"
|
||||
ba = "ba"
|
||||
ue = "ue"
|
||||
# file extensions that look like typos
|
||||
thm = "thm"
|
||||
# some typos we use for tests
|
||||
comparsion = "comparsion"
|
||||
modfiers = "modfiers"
|
||||
shcema = "shcema"
|
||||
|
||||
[default]
|
||||
extend-ignore-re = [
|
||||
|
||||
Binary file not shown.
|
After Width: | Height: | Size: 7.6 KiB |
Binary file not shown.
|
After Width: | Height: | Size: 7.0 KiB |
Binary file not shown.
|
After Width: | Height: | Size: 7.5 KiB |
+35
@@ -0,0 +1,35 @@
|
||||
{
|
||||
"$schema": "https://biomejs.dev/schemas/2.4.7/schema.json",
|
||||
"files": {
|
||||
"includes": ["packages/**/*.ts", "!packages/*/dist"],
|
||||
"ignoreUnknown": true
|
||||
},
|
||||
"formatter": {
|
||||
"enabled": true,
|
||||
"indentStyle": "space",
|
||||
"indentWidth": 2,
|
||||
"lineWidth": 90
|
||||
},
|
||||
"javascript": {
|
||||
"formatter": {
|
||||
"quoteStyle": "double",
|
||||
"trailingCommas": "all",
|
||||
"semicolons": "always"
|
||||
}
|
||||
},
|
||||
"linter": {
|
||||
"enabled": true,
|
||||
"rules": {
|
||||
"recommended": true,
|
||||
"style": {
|
||||
"noNonNullAssertion": "off"
|
||||
},
|
||||
"suspicious": {
|
||||
"noExplicitAny": "off"
|
||||
},
|
||||
"complexity": {
|
||||
"noForEach": "off"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,148 @@
|
||||
{
|
||||
"lockfileVersion": 1,
|
||||
"configVersion": 1,
|
||||
"workspaces": {
|
||||
"": {
|
||||
"devDependencies": {
|
||||
"@biomejs/biome": "^2.4.4",
|
||||
},
|
||||
},
|
||||
"packages/fff-bun": {
|
||||
"name": "@ff-labs/fff-bun",
|
||||
"version": "0.1.37",
|
||||
"bin": {
|
||||
"fff": "./scripts/cli.ts",
|
||||
"fff-demo": "./examples/search.ts",
|
||||
"fff-grep": "./examples/grep.ts",
|
||||
},
|
||||
"devDependencies": {
|
||||
"@types/bun": "^1.3.8",
|
||||
"typescript": "^5.0.0",
|
||||
},
|
||||
"optionalDependencies": {
|
||||
"@ff-labs/fff-bun-darwin-arm64": "0.0.0",
|
||||
"@ff-labs/fff-bun-darwin-x64": "0.0.0",
|
||||
"@ff-labs/fff-bun-linux-arm64-gnu": "0.0.0",
|
||||
"@ff-labs/fff-bun-linux-arm64-musl": "0.0.0",
|
||||
"@ff-labs/fff-bun-linux-x64-gnu": "0.0.0",
|
||||
"@ff-labs/fff-bun-linux-x64-musl": "0.0.0",
|
||||
"@ff-labs/fff-bun-win32-arm64": "0.0.0",
|
||||
"@ff-labs/fff-bun-win32-x64": "0.0.0",
|
||||
},
|
||||
"peerDependencies": {
|
||||
"bun": ">=1.0.0",
|
||||
},
|
||||
},
|
||||
"packages/fff-node": {
|
||||
"name": "@ff-labs/fff-node",
|
||||
"version": "0.1.37",
|
||||
"bin": {
|
||||
"fff-node": "./dist/scripts/cli.js",
|
||||
},
|
||||
"dependencies": {
|
||||
"ffi-rs": "^1.0.0",
|
||||
},
|
||||
"devDependencies": {
|
||||
"@types/node": "^22.0.0",
|
||||
"typescript": "^5.0.0",
|
||||
},
|
||||
"optionalDependencies": {
|
||||
"@ff-labs/fff-bun-darwin-arm64": "0.0.0",
|
||||
"@ff-labs/fff-bun-darwin-x64": "0.0.0",
|
||||
"@ff-labs/fff-bun-linux-arm64-gnu": "0.0.0",
|
||||
"@ff-labs/fff-bun-linux-arm64-musl": "0.0.0",
|
||||
"@ff-labs/fff-bun-linux-x64-gnu": "0.0.0",
|
||||
"@ff-labs/fff-bun-linux-x64-musl": "0.0.0",
|
||||
"@ff-labs/fff-bun-win32-arm64": "0.0.0",
|
||||
"@ff-labs/fff-bun-win32-x64": "0.0.0",
|
||||
},
|
||||
},
|
||||
},
|
||||
"packages": {
|
||||
"@biomejs/biome": ["@biomejs/biome@2.4.4", "", { "optionalDependencies": { "@biomejs/cli-darwin-arm64": "2.4.4", "@biomejs/cli-darwin-x64": "2.4.4", "@biomejs/cli-linux-arm64": "2.4.4", "@biomejs/cli-linux-arm64-musl": "2.4.4", "@biomejs/cli-linux-x64": "2.4.4", "@biomejs/cli-linux-x64-musl": "2.4.4", "@biomejs/cli-win32-arm64": "2.4.4", "@biomejs/cli-win32-x64": "2.4.4" }, "bin": { "biome": "bin/biome" } }, "sha512-tigwWS5KfJf0cABVd52NVaXyAVv4qpUXOWJ1rxFL8xF1RVoeS2q/LK+FHgYoKMclJCuRoCWAPy1IXaN9/mS61Q=="],
|
||||
|
||||
"@biomejs/cli-darwin-arm64": ["@biomejs/cli-darwin-arm64@2.4.4", "", { "os": "darwin", "cpu": "arm64" }, "sha512-jZ+Xc6qvD6tTH5jM6eKX44dcbyNqJHssfl2nnwT6vma6B1sj7ZLTGIk6N5QwVBs5xGN52r3trk5fgd3sQ9We9A=="],
|
||||
|
||||
"@biomejs/cli-darwin-x64": ["@biomejs/cli-darwin-x64@2.4.4", "", { "os": "darwin", "cpu": "x64" }, "sha512-Dh1a/+W+SUCXhEdL7TiX3ArPTFCQKJTI1mGncZNWfO+6suk+gYA4lNyJcBB+pwvF49uw0pEbUS49BgYOY4hzUg=="],
|
||||
|
||||
"@biomejs/cli-linux-arm64": ["@biomejs/cli-linux-arm64@2.4.4", "", { "os": "linux", "cpu": "arm64" }, "sha512-V/NFfbWhsUU6w+m5WYbBenlEAz8eYnSqRMDMAW3K+3v0tYVkNyZn8VU0XPxk/lOqNXLSCCrV7FmV/u3SjCBShg=="],
|
||||
|
||||
"@biomejs/cli-linux-arm64-musl": ["@biomejs/cli-linux-arm64-musl@2.4.4", "", { "os": "linux", "cpu": "arm64" }, "sha512-+sPAXq3bxmFwhVFJnSwkSF5Rw2ZAJMH3MF6C9IveAEOdSpgajPhoQhbbAK12SehN9j2QrHpk4J/cHsa/HqWaYQ=="],
|
||||
|
||||
"@biomejs/cli-linux-x64": ["@biomejs/cli-linux-x64@2.4.4", "", { "os": "linux", "cpu": "x64" }, "sha512-R4+ZCDtG9kHArasyBO+UBD6jr/FcFCTH8QkNTOCu0pRJzCWyWC4EtZa2AmUZB5h3e0jD7bRV2KvrENcf8rndBg=="],
|
||||
|
||||
"@biomejs/cli-linux-x64-musl": ["@biomejs/cli-linux-x64-musl@2.4.4", "", { "os": "linux", "cpu": "x64" }, "sha512-gGvFTGpOIQDb5CQ2VC0n9Z2UEqlP46c4aNgHmAMytYieTGEcfqhfCFnhs6xjt0S3igE6q5GLuIXtdQt3Izok+g=="],
|
||||
|
||||
"@biomejs/cli-win32-arm64": ["@biomejs/cli-win32-arm64@2.4.4", "", { "os": "win32", "cpu": "arm64" }, "sha512-trzCqM7x+Gn832zZHgr28JoYagQNX4CZkUZhMUac2YxvvyDRLJDrb5m9IA7CaZLlX6lTQmADVfLEKP1et1Ma4Q=="],
|
||||
|
||||
"@biomejs/cli-win32-x64": ["@biomejs/cli-win32-x64@2.4.4", "", { "os": "win32", "cpu": "x64" }, "sha512-gnOHKVPFAAPrpoPt2t+Q6FZ7RPry/FDV3GcpU53P3PtLNnQjBmKyN2Vh/JtqXet+H4pme8CC76rScwdjDcT1/A=="],
|
||||
|
||||
"@ff-labs/fff-bun": ["@ff-labs/fff-bun@workspace:packages/fff-bun"],
|
||||
|
||||
"@ff-labs/fff-node": ["@ff-labs/fff-node@workspace:packages/fff-node"],
|
||||
|
||||
"@oven/bun-darwin-aarch64": ["@oven/bun-darwin-aarch64@1.3.10", "", { "os": "darwin", "cpu": "arm64" }, "sha512-PXgg5gqcS/rHwa1hF0JdM1y5TiyejVrMHoBmWY/DjtfYZoFTXie1RCFOkoG0b5diOOmUcuYarMpH7CSNTqwj+w=="],
|
||||
|
||||
"@oven/bun-darwin-x64": ["@oven/bun-darwin-x64@1.3.10", "", { "os": "darwin", "cpu": "x64" }, "sha512-Nhssuh7GBpP5PiDSOl3+qnoIG7PJo+ec2oomDevnl9pRY6x6aD2gRt0JE+uf+A8Om2D6gjeHCxjEdrw5ZHE8mA=="],
|
||||
|
||||
"@oven/bun-darwin-x64-baseline": ["@oven/bun-darwin-x64-baseline@1.3.10", "", { "os": "darwin", "cpu": "x64" }, "sha512-w1gaTlqU0IJCmJ1X+PGHkdNU1n8Gemx5YKkjhkJIguvFINXEBB5U1KG82QsT65Tk4KyNMfbLTlmy4giAvUoKfA=="],
|
||||
|
||||
"@oven/bun-linux-aarch64": ["@oven/bun-linux-aarch64@1.3.10", "", { "os": "linux", "cpu": "arm64" }, "sha512-OUgPHfL6+PM2Q+tFZjcaycN3D7gdQdYlWnwMI31DXZKY1r4HINWk9aEz9t/rNaHg65edwNrt7dsv9TF7xK8xIA=="],
|
||||
|
||||
"@oven/bun-linux-aarch64-musl": ["@oven/bun-linux-aarch64-musl@1.3.10", "", { "os": "linux", "cpu": "arm64" }, "sha512-Ui5pAgM7JE9MzHokF0VglRMkbak3lTisY4Mf1AZutPACXWgKJC5aGrgnHBfkl7QS6fEeYb0juy1q4eRznRHOsw=="],
|
||||
|
||||
"@oven/bun-linux-x64": ["@oven/bun-linux-x64@1.3.10", "", { "os": "linux", "cpu": "x64" }, "sha512-bzUgYj/PIZziB/ZesIP9HUyfvh6Vlf3od+TrbTTyVEuCSMKzDPQVW/yEbRp0tcHO3alwiEXwJDrWrHAguXlgiQ=="],
|
||||
|
||||
"@oven/bun-linux-x64-baseline": ["@oven/bun-linux-x64-baseline@1.3.10", "", { "os": "linux", "cpu": "x64" }, "sha512-oqvMDYpX6dGJO03HgO5bXuccEsH3qbdO3MaAiAlO4CfkBPLUXz3N0DDElg5hz0L6ktdDVKbQVE5lfe+LAUISQg=="],
|
||||
|
||||
"@oven/bun-linux-x64-musl": ["@oven/bun-linux-x64-musl@1.3.10", "", { "os": "linux", "cpu": "x64" }, "sha512-poVXvOShekbexHq45b4MH/mRjQKwACAC8lHp3Tz/hEDuz0/20oncqScnmKwzhBPEpqJvydXficXfBYuSim8opw=="],
|
||||
|
||||
"@oven/bun-linux-x64-musl-baseline": ["@oven/bun-linux-x64-musl-baseline@1.3.10", "", { "os": "linux", "cpu": "x64" }, "sha512-/hOZ6S1VsTX6vtbhWVL9aAnOrdpuO54mAGUWpTdMz7dFG5UBZ/VUEiK0pBkq9A1rlBk0GeD/6Y4NBFl8Ha7cRA=="],
|
||||
|
||||
"@oven/bun-windows-aarch64": ["@oven/bun-windows-aarch64@1.3.10", "", { "os": "win32", "cpu": "arm64" }, "sha512-GXbz2swvN2DLw2dXZFeedMxSJtI64xQ9xp9Eg7Hjejg6mS2E4dP1xoQ2yAo2aZPi/2OBPAVaGzppI2q20XumHA=="],
|
||||
|
||||
"@oven/bun-windows-x64": ["@oven/bun-windows-x64@1.3.10", "", { "os": "win32", "cpu": "x64" }, "sha512-qaS1In3yfC/Z/IGQriVmF8GWwKuNqiw7feTSJWaQhH5IbL6ENR+4wGNPniZSJFaM/SKUO0e/YCRdoVBvgU4C1g=="],
|
||||
|
||||
"@oven/bun-windows-x64-baseline": ["@oven/bun-windows-x64-baseline@1.3.10", "", { "os": "win32", "cpu": "x64" }, "sha512-gh3UAHbUdDUG6fhLc1Csa4IGdtghue6U8oAIXWnUqawp6lwb3gOCRvp25IUnLF5vUHtgfMxuEUYV7YA2WxVutw=="],
|
||||
|
||||
"@types/bun": ["@types/bun@1.3.9", "", { "dependencies": { "bun-types": "1.3.9" } }, "sha512-KQ571yULOdWJiMH+RIWIOZ7B2RXQGpL1YQrBtLIV3FqDcCu6FsbFUBwhdKUlCKUpS3PJDsHlJ1QKlpxoVR+xtw=="],
|
||||
|
||||
"@types/node": ["@types/node@22.19.15", "", { "dependencies": { "undici-types": "~6.21.0" } }, "sha512-F0R/h2+dsy5wJAUe3tAU6oqa2qbWY5TpNfL/RGmo1y38hiyO1w3x2jPtt76wmuaJI4DQnOBu21cNXQ2STIUUWg=="],
|
||||
|
||||
"@yuuang/ffi-rs-android-arm64": ["@yuuang/ffi-rs-android-arm64@1.3.1", "", { "os": "android", "cpu": "arm64" }, "sha512-V4nmlXdOYZEa7GOxSExVG95SLp8FE0iTq2yKeN54UlfNMr3Sik+1Ff57LcCv7qYcn4TBqnBAt5rT3FAM6T6caQ=="],
|
||||
|
||||
"@yuuang/ffi-rs-darwin-arm64": ["@yuuang/ffi-rs-darwin-arm64@1.3.1", "", { "os": "darwin", "cpu": "arm64" }, "sha512-YlnTMIyzfW3mAULC5ZA774nzQfFlYXM0rrfq/8ZzWt+IMbYk55a++jrI+6JeKV+1EqlDS3TFBEFtjdBNG94KzQ=="],
|
||||
|
||||
"@yuuang/ffi-rs-darwin-x64": ["@yuuang/ffi-rs-darwin-x64@1.3.1", "", { "os": "darwin", "cpu": "x64" }, "sha512-sI3LpQQ34SX4nyOHc5yxA7FSqs9qPEUMqW/y/wWo9cuyPpaHMFsi/BeOVYsnC0syp3FrY7gzn6RnD6PlXCktXg=="],
|
||||
|
||||
"@yuuang/ffi-rs-linux-arm-gnueabihf": ["@yuuang/ffi-rs-linux-arm-gnueabihf@1.3.1", "", { "os": "linux", "cpu": "arm" }, "sha512-1WkcGkJTlwh4ZA59htKI+RXhiL3oKiYwLv7PO8LUf6FuADK73s5GcXp67iakKu243uYu+qGYr4RHco4ySddYhQ=="],
|
||||
|
||||
"@yuuang/ffi-rs-linux-arm64-gnu": ["@yuuang/ffi-rs-linux-arm64-gnu@1.3.1", "", { "os": "linux", "cpu": "arm64" }, "sha512-J2PwqviycZxaEVA0Bwv38LqGDGSB9A1DPN4iYginYJZSvTvKW8kh7Tis0HbZrX1YDKnY8hi3lt0N0tCTNPDH5Q=="],
|
||||
|
||||
"@yuuang/ffi-rs-linux-arm64-musl": ["@yuuang/ffi-rs-linux-arm64-musl@1.3.1", "", { "os": "linux", "cpu": "arm64" }, "sha512-Hn1W1hBPssTaqikU1Bqp1XUdDdOgbnYVIOtR++LVx66hhrtjf/xrIUQOhTm+NmOFDG16JUKXe1skfM4gpaqYwg=="],
|
||||
|
||||
"@yuuang/ffi-rs-linux-x64-gnu": ["@yuuang/ffi-rs-linux-x64-gnu@1.3.1", "", { "os": "linux", "cpu": "x64" }, "sha512-kW6e+oCYZPvpH2ppPsffA18e1aLowtmWTRjVlyHtY04g/nQDepQvDUkkcvInh9fW5jLna7PjHvktW1tVgYIj2A=="],
|
||||
|
||||
"@yuuang/ffi-rs-linux-x64-musl": ["@yuuang/ffi-rs-linux-x64-musl@1.3.1", "", { "os": "linux", "cpu": "x64" }, "sha512-HTwblAzruUS16nQPrez3ozvEHm1Xxh8J8w7rZYrpmAcNl1hzyOT8z/hY70M9Rt9fOqQ4Ovgor9qVy/U3ZJo0ZA=="],
|
||||
|
||||
"@yuuang/ffi-rs-win32-arm64-msvc": ["@yuuang/ffi-rs-win32-arm64-msvc@1.3.1", "", { "os": "win32", "cpu": "arm64" }, "sha512-WeZkGl2BP1U4tRhEQH+FXLQS52N8obp74smK5AAGOfzPAT1pHkq6+dVkC1QCSIt7dHJs7SPtlnQw+5DkdZYlWA=="],
|
||||
|
||||
"@yuuang/ffi-rs-win32-ia32-msvc": ["@yuuang/ffi-rs-win32-ia32-msvc@1.3.1", "", { "os": "win32", "cpu": [ "x64", "ia32", ] }, "sha512-rNGgMeCH5mdeHiMiJgt7wWXovZ+FHEfXhU9p4zZBH4n8M1/QnEsRUwlapISPLpILSGpoYS6iBuq9/fUlZY8Mhg=="],
|
||||
|
||||
"@yuuang/ffi-rs-win32-x64-msvc": ["@yuuang/ffi-rs-win32-x64-msvc@1.3.1", "", { "os": "win32", "cpu": "x64" }, "sha512-dr2LcLD2CXo2a7BktlOpV68QhayqiI112KxIJC9tBgQO/Dkdg4CPsdqmvzzLhFo64iC5RLl2BT7M5lJImrfUWw=="],
|
||||
|
||||
"bun": ["bun@1.3.10", "", { "optionalDependencies": { "@oven/bun-darwin-aarch64": "1.3.10", "@oven/bun-darwin-x64": "1.3.10", "@oven/bun-darwin-x64-baseline": "1.3.10", "@oven/bun-linux-aarch64": "1.3.10", "@oven/bun-linux-aarch64-musl": "1.3.10", "@oven/bun-linux-x64": "1.3.10", "@oven/bun-linux-x64-baseline": "1.3.10", "@oven/bun-linux-x64-musl": "1.3.10", "@oven/bun-linux-x64-musl-baseline": "1.3.10", "@oven/bun-windows-aarch64": "1.3.10", "@oven/bun-windows-x64": "1.3.10", "@oven/bun-windows-x64-baseline": "1.3.10" }, "os": [ "linux", "win32", "darwin", ], "cpu": [ "x64", "arm64", ], "bin": { "bun": "bin/bun.exe", "bunx": "bin/bunx.exe" } }, "sha512-S/CXaXXIyA4CMjdMkYQ4T2YMqnAn4s0ysD3mlsY4bUiOCqGlv28zck4Wd4H4kpvbekx15S9mUeLQ7Uxd0tYTLA=="],
|
||||
|
||||
"bun-types": ["bun-types@1.3.9", "", { "dependencies": { "@types/node": "*" } }, "sha512-+UBWWOakIP4Tswh0Bt0QD0alpTY8cb5hvgiYeWCMet9YukHbzuruIEeXC2D7nMJPB12kbh8C7XJykSexEqGKJg=="],
|
||||
|
||||
"ffi-rs": ["ffi-rs@1.3.1", "", { "optionalDependencies": { "@yuuang/ffi-rs-android-arm64": "1.3.1", "@yuuang/ffi-rs-darwin-arm64": "1.3.1", "@yuuang/ffi-rs-darwin-x64": "1.3.1", "@yuuang/ffi-rs-linux-arm-gnueabihf": "1.3.1", "@yuuang/ffi-rs-linux-arm64-gnu": "1.3.1", "@yuuang/ffi-rs-linux-arm64-musl": "1.3.1", "@yuuang/ffi-rs-linux-x64-gnu": "1.3.1", "@yuuang/ffi-rs-linux-x64-musl": "1.3.1", "@yuuang/ffi-rs-win32-arm64-msvc": "1.3.1", "@yuuang/ffi-rs-win32-ia32-msvc": "1.3.1", "@yuuang/ffi-rs-win32-x64-msvc": "1.3.1" } }, "sha512-ZyNXL9fnclnZV+waQmWB9JrfbIEyxQa1OWtMrHOrAgcC04PgP5hBMG5TdhVN8N4uT/eul8zCFMVnJUukAFFlXA=="],
|
||||
|
||||
"typescript": ["typescript@5.9.3", "", { "bin": { "tsc": "bin/tsc", "tsserver": "bin/tsserver" } }, "sha512-jl1vZzPDinLr9eUt3J/t7V6FgNEw9QjvBPdysz9KfQDD41fQrC2Y4vKQdiaUpFT4bXlb1RHhLpp8wtm6M5TgSw=="],
|
||||
|
||||
"undici-types": ["undici-types@6.21.0", "", {}, "sha512-iwDZqg0QAGrg9Rav5H4n0M64c3mkR59cJ6wQp+7C4nI0gsmExaedaYLNO44eT4AtBBwjbTiGPMlt2Md0T9H9JQ=="],
|
||||
|
||||
"bun-types/@types/node": ["@types/node@25.3.3", "", { "dependencies": { "undici-types": "~7.18.0" } }, "sha512-DpzbrH7wIcBaJibpKo9nnSQL0MTRdnWttGyE5haGwK86xgMOkFLp7vEyfQPGLOJh5wNYiJ3V9PmUMDhV9u8kkQ=="],
|
||||
|
||||
"bun-types/@types/node/undici-types": ["undici-types@7.18.2", "", {}, "sha512-AsuCzffGHJybSaRrmr5eHr81mwJU3kjw6M+uprWvCXiNeN9SOGwQ3Jn8jb8m3Z6izVgknn1R0FTCEAP2QrLY/w=="],
|
||||
}
|
||||
}
|
||||
@@ -1,20 +1,20 @@
|
||||
[package]
|
||||
name = "fff-c"
|
||||
version = "0.1.0"
|
||||
version = "0.8.1"
|
||||
edition = "2024"
|
||||
description = "C FFI bindings for fff-core - use from any language with C FFI support"
|
||||
description = "Raw C api of FFF file finder"
|
||||
license = "MIT"
|
||||
|
||||
[lib]
|
||||
crate-type = ["cdylib"]
|
||||
|
||||
[features]
|
||||
default = []
|
||||
zlob = ["fff/zlob"]
|
||||
|
||||
[dependencies]
|
||||
mimalloc.workspace = true
|
||||
once_cell.workspace = true
|
||||
tracing.workspace = true
|
||||
git2.workspace = true
|
||||
|
||||
fff-core = { path = "../fff-core" }
|
||||
fff-query-parser = { path = "../fff-query-parser" }
|
||||
serde = { version = "1.0", features = ["derive"] }
|
||||
fff = { package = "fff-search", path = "../fff-core" , version = "0.8.1" }
|
||||
fff-query-parser = { path = "../fff-query-parser" , version = "0.8.1" }
|
||||
serde_json = "1.0"
|
||||
|
||||
@@ -0,0 +1,28 @@
|
||||
language = "C"
|
||||
header = "/* Generated by cbindgen — do not edit manually. */"
|
||||
include_guard = "FFF_C_H"
|
||||
include_version = true
|
||||
no_includes = true
|
||||
sys_includes = ["stdint.h", "stdbool.h", "stddef.h"]
|
||||
|
||||
[export]
|
||||
include = [
|
||||
"FffResult",
|
||||
"FffSearchResult", "FffFileItem", "FffScore", "FffLocation",
|
||||
"FffGrepResult", "FffGrepMatch", "FffMatchRange",
|
||||
"FffScanProgress",
|
||||
]
|
||||
|
||||
[export.rename]
|
||||
"FffResult" = "FffResult"
|
||||
"FffSearchResult" = "FffSearchResult"
|
||||
"FffFileItem" = "FffFileItem"
|
||||
"FffScore" = "FffScore"
|
||||
"FffLocation" = "FffLocation"
|
||||
"FffGrepResult" = "FffGrepResult"
|
||||
"FffGrepMatch" = "FffGrepMatch"
|
||||
"FffMatchRange" = "FffMatchRange"
|
||||
"FffScanProgress" = "FffScanProgress"
|
||||
|
||||
[fn]
|
||||
sort_by = "None"
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,882 @@
|
||||
//! Stable accessor functions for `fff-c` FFI struct fields.
|
||||
//!
|
||||
//! # Why this exists
|
||||
//!
|
||||
//! `fff-c` exposes its result types as plain `#[repr(C)]` structs. External
|
||||
//! consumers (Emacs Lisp via `emacs-ffi`, Python `ctypes`, etc.) that access
|
||||
//! fields by hardcoding byte offsets break silently whenever the struct layout
|
||||
//! changes — a new field shifts every subsequent offset with no compile-time
|
||||
//! warning.
|
||||
//!
|
||||
//! These functions turn field access into a **stable named API**: callers bind
|
||||
//! to a symbol name once and are fully insulated from layout changes.
|
||||
//!
|
||||
//! # Usage from Emacs Lisp (example)
|
||||
//!
|
||||
//! ```elisp
|
||||
//! (define-ffi-function fff--grep-match-line-content
|
||||
//! "fff_grep_match_get_line_content" :pointer [:pointer] fff--library)
|
||||
//!
|
||||
//! (ffi-get-c-string (fff--grep-match-line-content match-ptr))
|
||||
//! ```
|
||||
//!
|
||||
//! # Array iteration
|
||||
//!
|
||||
//! To walk result arrays use `fff_search_result_get_item`,
|
||||
//! `fff_grep_result_get_match`, and `fff_search_result_get_score` — these are
|
||||
//! defined in the main `lib.rs` FFI surface alongside the search functions.
|
||||
|
||||
use std::ffi::c_char;
|
||||
use std::ptr;
|
||||
|
||||
use crate::ffi_types::{FffFileItem, FffGrepMatch, FffGrepResult, FffMatchRange, FffSearchResult};
|
||||
|
||||
// ── FffFileItem ──────────────────────────────────────────────────────────────
|
||||
|
||||
/// Returns the relative path of a file item (e.g. `"src/main.rs"`).
|
||||
///
|
||||
/// Returns null if `item` is null. The returned pointer is valid for the
|
||||
/// lifetime of the owning `FffSearchResult`; do not free it directly.
|
||||
///
|
||||
/// ## Safety
|
||||
/// `item` must be a valid `FffFileItem` pointer or null.
|
||||
#[unsafe(no_mangle)]
|
||||
pub unsafe extern "C" fn fff_file_item_get_relative_path(
|
||||
item: *const FffFileItem,
|
||||
) -> *const c_char {
|
||||
if item.is_null() {
|
||||
return ptr::null();
|
||||
}
|
||||
unsafe { (*item).relative_path }
|
||||
}
|
||||
|
||||
/// Returns the file-name component of a file item (e.g. `"main.rs"`).
|
||||
///
|
||||
/// Returns null if `item` is null. Do not free the returned pointer.
|
||||
///
|
||||
/// ## Safety
|
||||
/// `item` must be a valid `FffFileItem` pointer or null.
|
||||
#[unsafe(no_mangle)]
|
||||
pub unsafe extern "C" fn fff_file_item_get_file_name(item: *const FffFileItem) -> *const c_char {
|
||||
if item.is_null() {
|
||||
return ptr::null();
|
||||
}
|
||||
unsafe { (*item).file_name }
|
||||
}
|
||||
|
||||
/// Returns the git status string for a file item (e.g. `"M "`, `"??"`)
|
||||
/// or null if git is unavailable, the file is untracked, or `item` is null.
|
||||
///
|
||||
/// Do not free the returned pointer.
|
||||
///
|
||||
/// ## Safety
|
||||
/// `item` must be a valid `FffFileItem` pointer or null.
|
||||
#[unsafe(no_mangle)]
|
||||
pub unsafe extern "C" fn fff_file_item_get_git_status(item: *const FffFileItem) -> *const c_char {
|
||||
if item.is_null() {
|
||||
return ptr::null();
|
||||
}
|
||||
unsafe { (*item).git_status }
|
||||
}
|
||||
|
||||
/// Returns the file size in bytes. Returns `0` if `item` is null.
|
||||
///
|
||||
/// ## Safety
|
||||
/// `item` must be a valid `FffFileItem` pointer or null.
|
||||
#[unsafe(no_mangle)]
|
||||
pub unsafe extern "C" fn fff_file_item_get_size(item: *const FffFileItem) -> u64 {
|
||||
if item.is_null() {
|
||||
return 0;
|
||||
}
|
||||
unsafe { (*item).size }
|
||||
}
|
||||
|
||||
/// Returns the last-modified time as seconds since the UNIX epoch.
|
||||
/// Returns `0` if `item` is null.
|
||||
///
|
||||
/// ## Safety
|
||||
/// `item` must be a valid `FffFileItem` pointer or null.
|
||||
#[unsafe(no_mangle)]
|
||||
pub unsafe extern "C" fn fff_file_item_get_modified(item: *const FffFileItem) -> u64 {
|
||||
if item.is_null() {
|
||||
return 0;
|
||||
}
|
||||
unsafe { (*item).modified }
|
||||
}
|
||||
|
||||
/// Returns the combined frecency score. Returns `0` if `item` is null.
|
||||
///
|
||||
/// ## Safety
|
||||
/// `item` must be a valid `FffFileItem` pointer or null.
|
||||
#[unsafe(no_mangle)]
|
||||
pub unsafe extern "C" fn fff_file_item_get_total_frecency_score(item: *const FffFileItem) -> i64 {
|
||||
if item.is_null() {
|
||||
return 0;
|
||||
}
|
||||
unsafe { (*item).total_frecency_score }
|
||||
}
|
||||
|
||||
/// Returns the access-based frecency score. Returns `0` if `item` is null.
|
||||
///
|
||||
/// ## Safety
|
||||
/// `item` must be a valid `FffFileItem` pointer or null.
|
||||
#[unsafe(no_mangle)]
|
||||
pub unsafe extern "C" fn fff_file_item_get_access_frecency_score(item: *const FffFileItem) -> i64 {
|
||||
if item.is_null() {
|
||||
return 0;
|
||||
}
|
||||
unsafe { (*item).access_frecency_score }
|
||||
}
|
||||
|
||||
/// Returns the modification-based frecency score. Returns `0` if `item` is null.
|
||||
///
|
||||
/// ## Safety
|
||||
/// `item` must be a valid `FffFileItem` pointer or null.
|
||||
#[unsafe(no_mangle)]
|
||||
pub unsafe extern "C" fn fff_file_item_get_modification_frecency_score(
|
||||
item: *const FffFileItem,
|
||||
) -> i64 {
|
||||
if item.is_null() {
|
||||
return 0;
|
||||
}
|
||||
unsafe { (*item).modification_frecency_score }
|
||||
}
|
||||
|
||||
/// Returns `true` if the file was detected as binary. Returns `false` if `item` is null.
|
||||
///
|
||||
/// ## Safety
|
||||
/// `item` must be a valid `FffFileItem` pointer or null.
|
||||
#[unsafe(no_mangle)]
|
||||
pub unsafe extern "C" fn fff_file_item_get_is_binary(item: *const FffFileItem) -> bool {
|
||||
if item.is_null() {
|
||||
return false;
|
||||
}
|
||||
unsafe { (*item).is_binary }
|
||||
}
|
||||
|
||||
// ── FffGrepMatch ─────────────────────────────────────────────────────────────
|
||||
|
||||
/// Returns the relative path of the file containing this grep match.
|
||||
///
|
||||
/// Returns null if `m` is null. Do not free the returned pointer.
|
||||
///
|
||||
/// ## Safety
|
||||
/// `m` must be a valid `FffGrepMatch` pointer or null.
|
||||
#[unsafe(no_mangle)]
|
||||
pub unsafe extern "C" fn fff_grep_match_get_relative_path(m: *const FffGrepMatch) -> *const c_char {
|
||||
if m.is_null() {
|
||||
return ptr::null();
|
||||
}
|
||||
unsafe { (*m).relative_path }
|
||||
}
|
||||
|
||||
/// Returns the file-name component of the file containing this grep match.
|
||||
///
|
||||
/// Returns null if `m` is null. Do not free the returned pointer.
|
||||
///
|
||||
/// ## Safety
|
||||
/// `m` must be a valid `FffGrepMatch` pointer or null.
|
||||
#[unsafe(no_mangle)]
|
||||
pub unsafe extern "C" fn fff_grep_match_get_file_name(m: *const FffGrepMatch) -> *const c_char {
|
||||
if m.is_null() {
|
||||
return ptr::null();
|
||||
}
|
||||
unsafe { (*m).file_name }
|
||||
}
|
||||
|
||||
/// Returns the git status string for the matched file (e.g. `"M "`, `"??"`)
|
||||
/// or null if git is unavailable, the file is untracked, or `m` is null.
|
||||
///
|
||||
/// Do not free the returned pointer.
|
||||
///
|
||||
/// ## Safety
|
||||
/// `m` must be a valid `FffGrepMatch` pointer or null.
|
||||
#[unsafe(no_mangle)]
|
||||
pub unsafe extern "C" fn fff_grep_match_get_git_status(m: *const FffGrepMatch) -> *const c_char {
|
||||
if m.is_null() {
|
||||
return ptr::null();
|
||||
}
|
||||
unsafe { (*m).git_status }
|
||||
}
|
||||
|
||||
/// Returns the full text content of the matched line.
|
||||
///
|
||||
/// Returns null if `m` is null. Do not free the returned pointer.
|
||||
///
|
||||
/// ## Safety
|
||||
/// `m` must be a valid `FffGrepMatch` pointer or null.
|
||||
#[unsafe(no_mangle)]
|
||||
pub unsafe extern "C" fn fff_grep_match_get_line_content(m: *const FffGrepMatch) -> *const c_char {
|
||||
if m.is_null() {
|
||||
return ptr::null();
|
||||
}
|
||||
unsafe { (*m).line_content }
|
||||
}
|
||||
|
||||
/// Returns the 1-based line number of the match within its file.
|
||||
/// Returns `0` if `m` is null.
|
||||
///
|
||||
/// ## Safety
|
||||
/// `m` must be a valid `FffGrepMatch` pointer or null.
|
||||
#[unsafe(no_mangle)]
|
||||
pub unsafe extern "C" fn fff_grep_match_get_line_number(m: *const FffGrepMatch) -> u64 {
|
||||
if m.is_null() {
|
||||
return 0;
|
||||
}
|
||||
unsafe { (*m).line_number }
|
||||
}
|
||||
|
||||
/// Returns the 0-based column of the match start within its line.
|
||||
/// Returns `0` if `m` is null.
|
||||
///
|
||||
/// ## Safety
|
||||
/// `m` must be a valid `FffGrepMatch` pointer or null.
|
||||
#[unsafe(no_mangle)]
|
||||
pub unsafe extern "C" fn fff_grep_match_get_col(m: *const FffGrepMatch) -> u32 {
|
||||
if m.is_null() {
|
||||
return 0;
|
||||
}
|
||||
unsafe { (*m).col }
|
||||
}
|
||||
|
||||
/// Returns the byte offset of the match start from the beginning of the file.
|
||||
/// Returns `0` if `m` is null.
|
||||
///
|
||||
/// ## Safety
|
||||
/// `m` must be a valid `FffGrepMatch` pointer or null.
|
||||
#[unsafe(no_mangle)]
|
||||
pub unsafe extern "C" fn fff_grep_match_get_byte_offset(m: *const FffGrepMatch) -> u64 {
|
||||
if m.is_null() {
|
||||
return 0;
|
||||
}
|
||||
unsafe { (*m).byte_offset }
|
||||
}
|
||||
|
||||
/// Returns the file size in bytes for the matched file. Returns `0` if `m` is null.
|
||||
///
|
||||
/// ## Safety
|
||||
/// `m` must be a valid `FffGrepMatch` pointer or null.
|
||||
#[unsafe(no_mangle)]
|
||||
pub unsafe extern "C" fn fff_grep_match_get_size(m: *const FffGrepMatch) -> u64 {
|
||||
if m.is_null() {
|
||||
return 0;
|
||||
}
|
||||
unsafe { (*m).size }
|
||||
}
|
||||
|
||||
/// Returns the combined frecency score for the matched file.
|
||||
/// Returns `0` if `m` is null.
|
||||
///
|
||||
/// ## Safety
|
||||
/// `m` must be a valid `FffGrepMatch` pointer or null.
|
||||
#[unsafe(no_mangle)]
|
||||
pub unsafe extern "C" fn fff_grep_match_get_total_frecency_score(m: *const FffGrepMatch) -> i64 {
|
||||
if m.is_null() {
|
||||
return 0;
|
||||
}
|
||||
unsafe { (*m).total_frecency_score }
|
||||
}
|
||||
|
||||
/// Returns the access-based frecency score for the matched file.
|
||||
/// Returns `0` if `m` is null.
|
||||
///
|
||||
/// ## Safety
|
||||
/// `m` must be a valid `FffGrepMatch` pointer or null.
|
||||
#[unsafe(no_mangle)]
|
||||
pub unsafe extern "C" fn fff_grep_match_get_access_frecency_score(m: *const FffGrepMatch) -> i64 {
|
||||
if m.is_null() {
|
||||
return 0;
|
||||
}
|
||||
unsafe { (*m).access_frecency_score }
|
||||
}
|
||||
|
||||
/// Returns the modification-based frecency score for the matched file.
|
||||
/// Returns `0` if `m` is null.
|
||||
///
|
||||
/// ## Safety
|
||||
/// `m` must be a valid `FffGrepMatch` pointer or null.
|
||||
#[unsafe(no_mangle)]
|
||||
pub unsafe extern "C" fn fff_grep_match_get_modification_frecency_score(
|
||||
m: *const FffGrepMatch,
|
||||
) -> i64 {
|
||||
if m.is_null() {
|
||||
return 0;
|
||||
}
|
||||
unsafe { (*m).modification_frecency_score }
|
||||
}
|
||||
|
||||
/// Returns the last-modified time as seconds since the UNIX epoch for the matched file.
|
||||
/// Returns `0` if `m` is null.
|
||||
///
|
||||
/// ## Safety
|
||||
/// `m` must be a valid `FffGrepMatch` pointer or null.
|
||||
#[unsafe(no_mangle)]
|
||||
pub unsafe extern "C" fn fff_grep_match_get_modified(m: *const FffGrepMatch) -> u64 {
|
||||
if m.is_null() {
|
||||
return 0;
|
||||
}
|
||||
unsafe { (*m).modified }
|
||||
}
|
||||
|
||||
/// Returns the number of highlight ranges in this match. Returns `0` if `m` is null.
|
||||
///
|
||||
/// Use with [`fff_grep_match_get_match_range`] to iterate the highlight spans.
|
||||
///
|
||||
/// ## Safety
|
||||
/// `m` must be a valid `FffGrepMatch` pointer or null.
|
||||
#[unsafe(no_mangle)]
|
||||
pub unsafe extern "C" fn fff_grep_match_get_match_ranges_count(m: *const FffGrepMatch) -> u32 {
|
||||
if m.is_null() {
|
||||
return 0;
|
||||
}
|
||||
unsafe { (*m).match_ranges_count }
|
||||
}
|
||||
|
||||
/// Returns a pointer to the `index`-th [`FffMatchRange`] highlight span.
|
||||
///
|
||||
/// Returns null if `m` is null, `index >= match_ranges_count`, or the
|
||||
/// ranges array is null. The returned pointer is valid until the owning
|
||||
/// `FffGrepResult` is freed; do not free it directly.
|
||||
///
|
||||
/// ## Safety
|
||||
/// `m` must be a valid `FffGrepMatch` pointer or null.
|
||||
#[unsafe(no_mangle)]
|
||||
pub unsafe extern "C" fn fff_grep_match_get_match_range(
|
||||
m: *const FffGrepMatch,
|
||||
index: u32,
|
||||
) -> *const FffMatchRange {
|
||||
if m.is_null() {
|
||||
return ptr::null();
|
||||
}
|
||||
let m = unsafe { &*m };
|
||||
if index >= m.match_ranges_count || m.match_ranges.is_null() {
|
||||
return ptr::null();
|
||||
}
|
||||
unsafe { m.match_ranges.add(index as usize) }
|
||||
}
|
||||
|
||||
/// Returns the number of context lines captured before the match.
|
||||
/// Returns `0` if `m` is null.
|
||||
///
|
||||
/// Use with [`fff_grep_match_get_context_before`] to read each line.
|
||||
///
|
||||
/// ## Safety
|
||||
/// `m` must be a valid `FffGrepMatch` pointer or null.
|
||||
#[unsafe(no_mangle)]
|
||||
pub unsafe extern "C" fn fff_grep_match_get_context_before_count(m: *const FffGrepMatch) -> u32 {
|
||||
if m.is_null() {
|
||||
return 0;
|
||||
}
|
||||
unsafe { (*m).context_before_count }
|
||||
}
|
||||
|
||||
/// Returns the `index`-th context line before the match.
|
||||
///
|
||||
/// Returns null if `m` is null, `index >= context_before_count`, or the
|
||||
/// context array is null. Do not free the returned pointer.
|
||||
///
|
||||
/// ## Safety
|
||||
/// `m` must be a valid `FffGrepMatch` pointer or null.
|
||||
#[unsafe(no_mangle)]
|
||||
pub unsafe extern "C" fn fff_grep_match_get_context_before(
|
||||
m: *const FffGrepMatch,
|
||||
index: u32,
|
||||
) -> *const c_char {
|
||||
if m.is_null() {
|
||||
return ptr::null();
|
||||
}
|
||||
let m = unsafe { &*m };
|
||||
if index >= m.context_before_count || m.context_before.is_null() {
|
||||
return ptr::null();
|
||||
}
|
||||
unsafe { *m.context_before.add(index as usize) }
|
||||
}
|
||||
|
||||
/// Returns the number of context lines captured after the match.
|
||||
/// Returns `0` if `m` is null.
|
||||
///
|
||||
/// Use with [`fff_grep_match_get_context_after`] to read each line.
|
||||
///
|
||||
/// ## Safety
|
||||
/// `m` must be a valid `FffGrepMatch` pointer or null.
|
||||
#[unsafe(no_mangle)]
|
||||
pub unsafe extern "C" fn fff_grep_match_get_context_after_count(m: *const FffGrepMatch) -> u32 {
|
||||
if m.is_null() {
|
||||
return 0;
|
||||
}
|
||||
unsafe { (*m).context_after_count }
|
||||
}
|
||||
|
||||
/// Returns the `index`-th context line after the match.
|
||||
///
|
||||
/// Returns null if `m` is null, `index >= context_after_count`, or the
|
||||
/// context array is null. Do not free the returned pointer.
|
||||
///
|
||||
/// ## Safety
|
||||
/// `m` must be a valid `FffGrepMatch` pointer or null.
|
||||
#[unsafe(no_mangle)]
|
||||
pub unsafe extern "C" fn fff_grep_match_get_context_after(
|
||||
m: *const FffGrepMatch,
|
||||
index: u32,
|
||||
) -> *const c_char {
|
||||
if m.is_null() {
|
||||
return ptr::null();
|
||||
}
|
||||
let m = unsafe { &*m };
|
||||
if index >= m.context_after_count || m.context_after.is_null() {
|
||||
return ptr::null();
|
||||
}
|
||||
unsafe { *m.context_after.add(index as usize) }
|
||||
}
|
||||
|
||||
/// Returns the fuzzy match score. Returns `0` if `m` is null or no fuzzy
|
||||
/// score is present.
|
||||
///
|
||||
/// Always check [`fff_grep_match_get_has_fuzzy_score`] first; `0` is
|
||||
/// ambiguous without that flag.
|
||||
///
|
||||
/// ## Safety
|
||||
/// `m` must be a valid `FffGrepMatch` pointer or null.
|
||||
#[unsafe(no_mangle)]
|
||||
pub unsafe extern "C" fn fff_grep_match_get_fuzzy_score(m: *const FffGrepMatch) -> u16 {
|
||||
if m.is_null() {
|
||||
return 0;
|
||||
}
|
||||
unsafe { (*m).fuzzy_score }
|
||||
}
|
||||
|
||||
/// Returns `true` if this match carries a valid fuzzy score.
|
||||
/// Returns `false` if `m` is null.
|
||||
///
|
||||
/// ## Safety
|
||||
/// `m` must be a valid `FffGrepMatch` pointer or null.
|
||||
#[unsafe(no_mangle)]
|
||||
pub unsafe extern "C" fn fff_grep_match_get_has_fuzzy_score(m: *const FffGrepMatch) -> bool {
|
||||
if m.is_null() {
|
||||
return false;
|
||||
}
|
||||
unsafe { (*m).has_fuzzy_score }
|
||||
}
|
||||
|
||||
/// Returns `true` if the match was identified as a symbol definition.
|
||||
/// Returns `false` if `m` is null.
|
||||
///
|
||||
/// ## Safety
|
||||
/// `m` must be a valid `FffGrepMatch` pointer or null.
|
||||
#[unsafe(no_mangle)]
|
||||
pub unsafe extern "C" fn fff_grep_match_get_is_definition(m: *const FffGrepMatch) -> bool {
|
||||
if m.is_null() {
|
||||
return false;
|
||||
}
|
||||
unsafe { (*m).is_definition }
|
||||
}
|
||||
|
||||
/// Returns `true` if the matched file was detected as binary.
|
||||
/// Returns `false` if `m` is null.
|
||||
///
|
||||
/// ## Safety
|
||||
/// `m` must be a valid `FffGrepMatch` pointer or null.
|
||||
#[unsafe(no_mangle)]
|
||||
pub unsafe extern "C" fn fff_grep_match_get_is_binary(m: *const FffGrepMatch) -> bool {
|
||||
if m.is_null() {
|
||||
return false;
|
||||
}
|
||||
unsafe { (*m).is_binary }
|
||||
}
|
||||
|
||||
// ── FffSearchResult ──────────────────────────────────────────────────────────
|
||||
|
||||
/// Returns the number of items in the result. Returns `0` if `r` is null.
|
||||
///
|
||||
/// ## Safety
|
||||
/// `r` must be a valid `FffSearchResult` pointer or null.
|
||||
#[unsafe(no_mangle)]
|
||||
pub unsafe extern "C" fn fff_search_result_get_count(r: *const FffSearchResult) -> u32 {
|
||||
if r.is_null() {
|
||||
return 0;
|
||||
}
|
||||
unsafe { (*r).count }
|
||||
}
|
||||
|
||||
/// Returns the total number of files that matched before the result was
|
||||
/// truncated to the page size. Returns `0` if `r` is null.
|
||||
///
|
||||
/// ## Safety
|
||||
/// `r` must be a valid `FffSearchResult` pointer or null.
|
||||
#[unsafe(no_mangle)]
|
||||
pub unsafe extern "C" fn fff_search_result_get_total_matched(r: *const FffSearchResult) -> u32 {
|
||||
if r.is_null() {
|
||||
return 0;
|
||||
}
|
||||
unsafe { (*r).total_matched }
|
||||
}
|
||||
|
||||
/// Returns the total number of indexed files considered during search.
|
||||
/// Returns `0` if `r` is null.
|
||||
///
|
||||
/// ## Safety
|
||||
/// `r` must be a valid `FffSearchResult` pointer or null.
|
||||
#[unsafe(no_mangle)]
|
||||
pub unsafe extern "C" fn fff_search_result_get_total_files(r: *const FffSearchResult) -> u32 {
|
||||
if r.is_null() {
|
||||
return 0;
|
||||
}
|
||||
unsafe { (*r).total_files }
|
||||
}
|
||||
|
||||
// ── FffGrepResult ─────────────────────────────────────────────────────────────
|
||||
|
||||
/// Returns the number of matches in the result. Returns `0` if `r` is null.
|
||||
///
|
||||
/// ## Safety
|
||||
/// `r` must be a valid `FffGrepResult` pointer or null.
|
||||
#[unsafe(no_mangle)]
|
||||
pub unsafe extern "C" fn fff_grep_result_get_count(r: *const FffGrepResult) -> u32 {
|
||||
if r.is_null() {
|
||||
return 0;
|
||||
}
|
||||
unsafe { (*r).count }
|
||||
}
|
||||
|
||||
/// Returns the total number of matches found across all pages.
|
||||
/// Returns `0` if `r` is null.
|
||||
///
|
||||
/// ## Safety
|
||||
/// `r` must be a valid `FffGrepResult` pointer or null.
|
||||
#[unsafe(no_mangle)]
|
||||
pub unsafe extern "C" fn fff_grep_result_get_total_matched(r: *const FffGrepResult) -> u32 {
|
||||
if r.is_null() {
|
||||
return 0;
|
||||
}
|
||||
unsafe { (*r).total_matched }
|
||||
}
|
||||
|
||||
/// Returns the number of files actually opened and searched in this call.
|
||||
/// Returns `0` if `r` is null.
|
||||
///
|
||||
/// ## Safety
|
||||
/// `r` must be a valid `FffGrepResult` pointer or null.
|
||||
#[unsafe(no_mangle)]
|
||||
pub unsafe extern "C" fn fff_grep_result_get_total_files_searched(r: *const FffGrepResult) -> u32 {
|
||||
if r.is_null() {
|
||||
return 0;
|
||||
}
|
||||
unsafe { (*r).total_files_searched }
|
||||
}
|
||||
|
||||
/// Returns the total number of indexed files before any filtering.
|
||||
/// Returns `0` if `r` is null.
|
||||
///
|
||||
/// ## Safety
|
||||
/// `r` must be a valid `FffGrepResult` pointer or null.
|
||||
#[unsafe(no_mangle)]
|
||||
pub unsafe extern "C" fn fff_grep_result_get_total_files(r: *const FffGrepResult) -> u32 {
|
||||
if r.is_null() {
|
||||
return 0;
|
||||
}
|
||||
unsafe { (*r).total_files }
|
||||
}
|
||||
|
||||
/// Returns the number of files eligible for search after path/type filtering.
|
||||
/// Returns `0` if `r` is null.
|
||||
///
|
||||
/// ## Safety
|
||||
/// `r` must be a valid `FffGrepResult` pointer or null.
|
||||
#[unsafe(no_mangle)]
|
||||
pub unsafe extern "C" fn fff_grep_result_get_filtered_file_count(r: *const FffGrepResult) -> u32 {
|
||||
if r.is_null() {
|
||||
return 0;
|
||||
}
|
||||
unsafe { (*r).filtered_file_count }
|
||||
}
|
||||
|
||||
/// Returns the file offset for the next page, or `0` if all files have been
|
||||
/// searched or `r` is null. Pass this value as `file_offset` to a subsequent
|
||||
/// `fff_live_grep` or `fff_multi_grep` call to continue pagination.
|
||||
///
|
||||
/// ## Safety
|
||||
/// `r` must be a valid `FffGrepResult` pointer or null.
|
||||
#[unsafe(no_mangle)]
|
||||
pub unsafe extern "C" fn fff_grep_result_get_next_file_offset(r: *const FffGrepResult) -> u32 {
|
||||
if r.is_null() {
|
||||
return 0;
|
||||
}
|
||||
unsafe { (*r).next_file_offset }
|
||||
}
|
||||
|
||||
/// Returns the regex compilation error string if the engine fell back to
|
||||
/// literal matching, or null if there was no error or `r` is null.
|
||||
///
|
||||
/// Do not free the returned pointer.
|
||||
///
|
||||
/// ## Safety
|
||||
/// `r` must be a valid `FffGrepResult` pointer or null.
|
||||
#[unsafe(no_mangle)]
|
||||
pub unsafe extern "C" fn fff_grep_result_get_regex_fallback_error(
|
||||
r: *const FffGrepResult,
|
||||
) -> *const c_char {
|
||||
if r.is_null() {
|
||||
return ptr::null();
|
||||
}
|
||||
unsafe { (*r).regex_fallback_error }
|
||||
}
|
||||
|
||||
// ── Tests ─────────────────────────────────────────────────────────────────────
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use std::ffi::CString;
|
||||
use std::ptr;
|
||||
|
||||
// ── helpers ──────────────────────────────────────────────────────────────
|
||||
|
||||
fn make_file_item(path: &str, name: &str) -> FffFileItem {
|
||||
FffFileItem {
|
||||
relative_path: CString::new(path).unwrap().into_raw(),
|
||||
file_name: CString::new(name).unwrap().into_raw(),
|
||||
git_status: ptr::null_mut(),
|
||||
size: 1024,
|
||||
modified: 1_700_000_000,
|
||||
access_frecency_score: 10,
|
||||
modification_frecency_score: 20,
|
||||
total_frecency_score: 30,
|
||||
is_binary: false,
|
||||
}
|
||||
}
|
||||
|
||||
unsafe fn free_file_item(item: &mut FffFileItem) {
|
||||
unsafe {
|
||||
if !item.relative_path.is_null() {
|
||||
drop(CString::from_raw(item.relative_path));
|
||||
}
|
||||
if !item.file_name.is_null() {
|
||||
drop(CString::from_raw(item.file_name));
|
||||
}
|
||||
if !item.git_status.is_null() {
|
||||
drop(CString::from_raw(item.git_status));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn make_grep_match(path: &str, line: &str) -> FffGrepMatch {
|
||||
FffGrepMatch {
|
||||
relative_path: CString::new(path).unwrap().into_raw(),
|
||||
file_name: CString::new("file.rs").unwrap().into_raw(),
|
||||
git_status: ptr::null_mut(),
|
||||
line_content: CString::new(line).unwrap().into_raw(),
|
||||
match_ranges: ptr::null_mut(),
|
||||
context_before: ptr::null_mut(),
|
||||
context_after: ptr::null_mut(),
|
||||
size: 512,
|
||||
modified: 1_600_000_000,
|
||||
total_frecency_score: 5,
|
||||
access_frecency_score: 6,
|
||||
modification_frecency_score: 7,
|
||||
line_number: 42,
|
||||
byte_offset: 100,
|
||||
col: 8,
|
||||
match_ranges_count: 0,
|
||||
context_before_count: 0,
|
||||
context_after_count: 0,
|
||||
fuzzy_score: 0,
|
||||
has_fuzzy_score: false,
|
||||
is_binary: false,
|
||||
is_definition: true,
|
||||
}
|
||||
}
|
||||
|
||||
unsafe fn free_grep_match(m: &mut FffGrepMatch) {
|
||||
unsafe {
|
||||
if !m.relative_path.is_null() {
|
||||
drop(CString::from_raw(m.relative_path));
|
||||
}
|
||||
if !m.file_name.is_null() {
|
||||
drop(CString::from_raw(m.file_name));
|
||||
}
|
||||
if !m.line_content.is_null() {
|
||||
drop(CString::from_raw(m.line_content));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn make_search_result(count: u32, total: u32, files: u32) -> FffSearchResult {
|
||||
FffSearchResult {
|
||||
items: ptr::null_mut(),
|
||||
scores: ptr::null_mut(),
|
||||
count,
|
||||
total_matched: total,
|
||||
total_files: files,
|
||||
location: crate::ffi_types::FffLocation {
|
||||
tag: 0,
|
||||
line: 0,
|
||||
col: 0,
|
||||
end_line: 0,
|
||||
end_col: 0,
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
fn make_grep_result() -> FffGrepResult {
|
||||
FffGrepResult {
|
||||
items: ptr::null_mut(),
|
||||
count: 3,
|
||||
total_matched: 10,
|
||||
total_files_searched: 50,
|
||||
total_files: 200,
|
||||
filtered_file_count: 80,
|
||||
next_file_offset: 51,
|
||||
regex_fallback_error: ptr::null_mut(),
|
||||
}
|
||||
}
|
||||
|
||||
// ── null-guard tests: every function returns its zero-value on NULL ───────
|
||||
|
||||
#[test]
|
||||
fn null_file_item_returns_null_or_zero() {
|
||||
let null: *const FffFileItem = ptr::null();
|
||||
unsafe {
|
||||
assert!(fff_file_item_get_relative_path(null).is_null());
|
||||
assert!(fff_file_item_get_file_name(null).is_null());
|
||||
assert!(fff_file_item_get_git_status(null).is_null());
|
||||
assert_eq!(fff_file_item_get_size(null), 0);
|
||||
assert_eq!(fff_file_item_get_modified(null), 0);
|
||||
assert_eq!(fff_file_item_get_access_frecency_score(null), 0);
|
||||
assert_eq!(fff_file_item_get_modification_frecency_score(null), 0);
|
||||
assert_eq!(fff_file_item_get_total_frecency_score(null), 0);
|
||||
assert!(!fff_file_item_get_is_binary(null));
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn null_grep_match_returns_null_or_zero() {
|
||||
let null: *const FffGrepMatch = ptr::null();
|
||||
unsafe {
|
||||
assert!(fff_grep_match_get_relative_path(null).is_null());
|
||||
assert!(fff_grep_match_get_file_name(null).is_null());
|
||||
assert!(fff_grep_match_get_git_status(null).is_null());
|
||||
assert!(fff_grep_match_get_line_content(null).is_null());
|
||||
assert_eq!(fff_grep_match_get_line_number(null), 0);
|
||||
assert_eq!(fff_grep_match_get_byte_offset(null), 0);
|
||||
assert_eq!(fff_grep_match_get_col(null), 0);
|
||||
assert_eq!(fff_grep_match_get_size(null), 0);
|
||||
assert_eq!(fff_grep_match_get_modified(null), 0);
|
||||
assert_eq!(fff_grep_match_get_total_frecency_score(null), 0);
|
||||
assert_eq!(fff_grep_match_get_access_frecency_score(null), 0);
|
||||
assert_eq!(fff_grep_match_get_modification_frecency_score(null), 0);
|
||||
assert_eq!(fff_grep_match_get_match_ranges_count(null), 0);
|
||||
assert_eq!(fff_grep_match_get_context_before_count(null), 0);
|
||||
assert_eq!(fff_grep_match_get_context_after_count(null), 0);
|
||||
assert!(!fff_grep_match_get_has_fuzzy_score(null));
|
||||
assert_eq!(fff_grep_match_get_fuzzy_score(null), 0);
|
||||
assert!(!fff_grep_match_get_is_binary(null));
|
||||
assert!(!fff_grep_match_get_is_definition(null));
|
||||
assert!(fff_grep_match_get_context_before(null, 0).is_null());
|
||||
assert!(fff_grep_match_get_context_after(null, 0).is_null());
|
||||
assert!(fff_grep_match_get_match_range(null, 0).is_null());
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn null_search_result_returns_zero() {
|
||||
let null: *const FffSearchResult = ptr::null();
|
||||
unsafe {
|
||||
assert_eq!(fff_search_result_get_count(null), 0);
|
||||
assert_eq!(fff_search_result_get_total_matched(null), 0);
|
||||
assert_eq!(fff_search_result_get_total_files(null), 0);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn null_grep_result_returns_zero_or_null() {
|
||||
let null: *const FffGrepResult = ptr::null();
|
||||
unsafe {
|
||||
assert_eq!(fff_grep_result_get_count(null), 0);
|
||||
assert_eq!(fff_grep_result_get_total_matched(null), 0);
|
||||
assert_eq!(fff_grep_result_get_total_files_searched(null), 0);
|
||||
assert_eq!(fff_grep_result_get_total_files(null), 0);
|
||||
assert_eq!(fff_grep_result_get_filtered_file_count(null), 0);
|
||||
assert_eq!(fff_grep_result_get_next_file_offset(null), 0);
|
||||
assert!(fff_grep_result_get_regex_fallback_error(null).is_null());
|
||||
}
|
||||
}
|
||||
|
||||
// ── data correctness tests ────────────────────────────────────────────────
|
||||
|
||||
#[test]
|
||||
fn file_item_getters_return_correct_values() {
|
||||
let mut item = make_file_item("src/main.rs", "main.rs");
|
||||
let p = &item as *const FffFileItem;
|
||||
unsafe {
|
||||
let path = std::ffi::CStr::from_ptr(fff_file_item_get_relative_path(p));
|
||||
assert_eq!(path.to_str().unwrap(), "src/main.rs");
|
||||
|
||||
let name = std::ffi::CStr::from_ptr(fff_file_item_get_file_name(p));
|
||||
assert_eq!(name.to_str().unwrap(), "main.rs");
|
||||
|
||||
assert!(fff_file_item_get_git_status(p).is_null());
|
||||
assert_eq!(fff_file_item_get_size(p), 1024);
|
||||
assert_eq!(fff_file_item_get_modified(p), 1_700_000_000);
|
||||
assert_eq!(fff_file_item_get_access_frecency_score(p), 10);
|
||||
assert_eq!(fff_file_item_get_modification_frecency_score(p), 20);
|
||||
assert_eq!(fff_file_item_get_total_frecency_score(p), 30);
|
||||
assert!(!fff_file_item_get_is_binary(p));
|
||||
|
||||
free_file_item(&mut item);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn grep_match_getters_return_correct_values() {
|
||||
let mut m = make_grep_match("src/lib.rs", "fn hello()");
|
||||
let p = &m as *const FffGrepMatch;
|
||||
unsafe {
|
||||
let path = std::ffi::CStr::from_ptr(fff_grep_match_get_relative_path(p));
|
||||
assert_eq!(path.to_str().unwrap(), "src/lib.rs");
|
||||
|
||||
let line = std::ffi::CStr::from_ptr(fff_grep_match_get_line_content(p));
|
||||
assert_eq!(line.to_str().unwrap(), "fn hello()");
|
||||
|
||||
assert_eq!(fff_grep_match_get_line_number(p), 42);
|
||||
assert_eq!(fff_grep_match_get_byte_offset(p), 100);
|
||||
assert_eq!(fff_grep_match_get_col(p), 8);
|
||||
assert_eq!(fff_grep_match_get_size(p), 512);
|
||||
assert_eq!(fff_grep_match_get_modified(p), 1_600_000_000);
|
||||
assert_eq!(fff_grep_match_get_total_frecency_score(p), 5);
|
||||
assert_eq!(fff_grep_match_get_access_frecency_score(p), 6);
|
||||
assert_eq!(fff_grep_match_get_modification_frecency_score(p), 7);
|
||||
assert_eq!(fff_grep_match_get_match_ranges_count(p), 0);
|
||||
assert!(!fff_grep_match_get_has_fuzzy_score(p));
|
||||
assert!(!fff_grep_match_get_is_binary(p));
|
||||
assert!(fff_grep_match_get_is_definition(p));
|
||||
|
||||
free_grep_match(&mut m);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn search_result_getters_return_correct_values() {
|
||||
let r = make_search_result(5, 20, 100);
|
||||
let p = &r as *const FffSearchResult;
|
||||
unsafe {
|
||||
assert_eq!(fff_search_result_get_count(p), 5);
|
||||
assert_eq!(fff_search_result_get_total_matched(p), 20);
|
||||
assert_eq!(fff_search_result_get_total_files(p), 100);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn grep_result_getters_return_correct_values() {
|
||||
let r = make_grep_result();
|
||||
let p = &r as *const FffGrepResult;
|
||||
unsafe {
|
||||
assert_eq!(fff_grep_result_get_count(p), 3);
|
||||
assert_eq!(fff_grep_result_get_total_matched(p), 10);
|
||||
assert_eq!(fff_grep_result_get_total_files_searched(p), 50);
|
||||
assert_eq!(fff_grep_result_get_total_files(p), 200);
|
||||
assert_eq!(fff_grep_result_get_filtered_file_count(p), 80);
|
||||
assert_eq!(fff_grep_result_get_next_file_offset(p), 51);
|
||||
assert!(fff_grep_result_get_regex_fallback_error(p).is_null());
|
||||
}
|
||||
}
|
||||
}
|
||||
+662
-158
@@ -1,129 +1,117 @@
|
||||
//! FFI-compatible type definitions
|
||||
//!
|
||||
//! These types use #[repr(C)] for C ABI compatibility and implement
|
||||
//! serde traits for JSON serialization.
|
||||
//! All result types use `#[repr(C)]` structs for direct memory access from any
|
||||
//! language with C FFI support. No JSON serialization is used for search or grep
|
||||
//! results — callers read struct fields directly.
|
||||
|
||||
use std::ffi::{CString, c_char};
|
||||
use std::ffi::{CString, c_char, c_void};
|
||||
use std::ptr;
|
||||
|
||||
use fff_core::git::format_git_status;
|
||||
use fff_core::{FileItem, Location, Score, SearchResult};
|
||||
use serde::{Deserialize, Serialize};
|
||||
use fff::file_picker::FilePicker;
|
||||
use fff::git::format_git_status;
|
||||
use fff::{
|
||||
DirItem, DirSearchResult, FileItem, GrepMatch, GrepResult, Location, MixedItemRef,
|
||||
MixedSearchResult, Score, SearchResult,
|
||||
};
|
||||
|
||||
/// Result type returned by all FFI functions
|
||||
/// Returned as a heap-allocated pointer that must be freed with fff_free_result
|
||||
/// Allocate a heap CString from a `&str`, returning a raw pointer.
|
||||
fn cstring_new(s: &str) -> *mut c_char {
|
||||
CString::new(s).unwrap_or_default().into_raw()
|
||||
}
|
||||
|
||||
/// Convert a `Vec<T>` into a raw pointer + count, leaking the memory.
|
||||
fn vec_to_raw<T>(v: Vec<T>) -> (*mut T, u32) {
|
||||
if v.is_empty() {
|
||||
return (ptr::null_mut(), 0);
|
||||
}
|
||||
let count = v.len() as u32;
|
||||
let mut boxed = v.into_boxed_slice();
|
||||
let p = boxed.as_mut_ptr();
|
||||
std::mem::forget(boxed);
|
||||
(p, count)
|
||||
}
|
||||
|
||||
/// Convert a `&[String]` into a heap-allocated array of C strings.
|
||||
fn strings_to_raw(v: &[String]) -> (*mut *mut c_char, u32) {
|
||||
if v.is_empty() {
|
||||
return (ptr::null_mut(), 0);
|
||||
}
|
||||
let ptrs: Vec<*mut c_char> = v.iter().map(|s| cstring_new(s)).collect();
|
||||
vec_to_raw(ptrs)
|
||||
}
|
||||
|
||||
/// Free a heap-allocated array of C strings.
|
||||
///
|
||||
/// ## Safety
|
||||
/// `arr` must have been produced by `strings_to_raw`.
|
||||
unsafe fn free_cstring_array(arr: *mut *mut c_char, count: u32) {
|
||||
if arr.is_null() {
|
||||
return;
|
||||
}
|
||||
unsafe {
|
||||
let ptrs = Vec::from_raw_parts(arr, count as usize, count as usize);
|
||||
for p in ptrs {
|
||||
if !p.is_null() {
|
||||
drop(CString::from_raw(p));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// A file item returned by `fff_search`.
|
||||
///
|
||||
/// All string fields are heap-allocated and owned by the parent `FffSearchResult`.
|
||||
/// Free the entire result with `fff_free_search_result`.
|
||||
#[repr(C)]
|
||||
pub struct FffResult {
|
||||
/// Whether the operation succeeded
|
||||
pub success: bool,
|
||||
/// JSON data on success (null-terminated string, caller must free)
|
||||
pub data: *mut c_char,
|
||||
/// Error message on failure (null-terminated string, caller must free)
|
||||
pub error: *mut c_char,
|
||||
}
|
||||
|
||||
impl FffResult {
|
||||
/// Create a successful result with no data, returned as heap pointer
|
||||
pub fn ok_empty() -> *mut Self {
|
||||
Box::into_raw(Box::new(FffResult {
|
||||
success: true,
|
||||
data: ptr::null_mut(),
|
||||
error: ptr::null_mut(),
|
||||
}))
|
||||
}
|
||||
|
||||
/// Create a successful result with data, returned as heap pointer
|
||||
pub fn ok_data(data: &str) -> *mut Self {
|
||||
Box::into_raw(Box::new(FffResult {
|
||||
success: true,
|
||||
data: CString::new(data).unwrap_or_default().into_raw(),
|
||||
error: ptr::null_mut(),
|
||||
}))
|
||||
}
|
||||
|
||||
/// Create an error result, returned as heap pointer
|
||||
pub fn err(error: &str) -> *mut Self {
|
||||
Box::into_raw(Box::new(FffResult {
|
||||
success: false,
|
||||
data: ptr::null_mut(),
|
||||
error: CString::new(error).unwrap_or_default().into_raw(),
|
||||
}))
|
||||
}
|
||||
}
|
||||
|
||||
/// Initialization options (JSON-deserializable)
|
||||
#[derive(Debug, Deserialize)]
|
||||
pub struct InitOptions {
|
||||
/// Base directory to index (required)
|
||||
pub base_path: String,
|
||||
/// Path to frecency database (optional, omit to skip frecency initialization)
|
||||
pub frecency_db_path: Option<String>,
|
||||
/// Path to query history database (optional, omit to skip query tracker initialization)
|
||||
pub history_db_path: Option<String>,
|
||||
/// Use unsafe no-lock mode for databases (optional, defaults to false)
|
||||
#[serde(default)]
|
||||
pub use_unsafe_no_lock: bool,
|
||||
}
|
||||
|
||||
/// Search options (JSON-deserializable)
|
||||
#[derive(Debug, Default, Deserialize)]
|
||||
pub struct SearchOptions {
|
||||
/// Maximum threads for parallel search (0 = auto)
|
||||
pub max_threads: Option<usize>,
|
||||
/// Current file path (for deprioritization)
|
||||
pub current_file: Option<String>,
|
||||
/// Combo boost score multiplier
|
||||
pub combo_boost_multiplier: Option<i32>,
|
||||
/// Minimum combo count for boost
|
||||
pub min_combo_count: Option<u32>,
|
||||
/// Page index for pagination
|
||||
pub page_index: Option<usize>,
|
||||
/// Page size for pagination
|
||||
pub page_size: Option<usize>,
|
||||
}
|
||||
|
||||
/// Scan progress (JSON-serializable)
|
||||
#[derive(Debug, Serialize)]
|
||||
pub struct ScanProgress {
|
||||
pub scanned_files_count: usize,
|
||||
pub is_scanning: bool,
|
||||
}
|
||||
|
||||
/// File item for JSON serialization
|
||||
#[derive(Debug, Serialize)]
|
||||
pub struct FileItemJson {
|
||||
pub path: String,
|
||||
pub relative_path: String,
|
||||
pub file_name: String,
|
||||
pub struct FffFileItem {
|
||||
pub relative_path: *mut c_char,
|
||||
pub file_name: *mut c_char,
|
||||
pub git_status: *mut c_char,
|
||||
pub size: u64,
|
||||
pub modified: u64,
|
||||
pub access_frecency_score: i64,
|
||||
pub modification_frecency_score: i64,
|
||||
pub total_frecency_score: i64,
|
||||
pub git_status: String,
|
||||
pub is_binary: bool,
|
||||
}
|
||||
|
||||
impl FileItemJson {
|
||||
pub fn from_file_item(item: &FileItem) -> Self {
|
||||
FileItemJson {
|
||||
path: item.path.to_string_lossy().to_string(),
|
||||
relative_path: item.relative_path.clone(),
|
||||
file_name: item.file_name.clone(),
|
||||
impl FffFileItem {
|
||||
pub fn from_item(item: &FileItem, picker: &FilePicker) -> Self {
|
||||
FffFileItem {
|
||||
relative_path: cstring_new(&item.relative_path(picker)),
|
||||
file_name: cstring_new(&item.file_name(picker)),
|
||||
git_status: cstring_new(format_git_status(item.git_status)),
|
||||
size: item.size,
|
||||
modified: item.modified,
|
||||
access_frecency_score: item.access_frecency_score,
|
||||
modification_frecency_score: item.modification_frecency_score,
|
||||
total_frecency_score: item.total_frecency_score,
|
||||
git_status: format_git_status(item.git_status).to_string(),
|
||||
is_binary: item.is_binary,
|
||||
access_frecency_score: item.access_frecency_score as i64,
|
||||
modification_frecency_score: item.modification_frecency_score as i64,
|
||||
total_frecency_score: item.total_frecency_score() as i64,
|
||||
is_binary: item.is_binary(),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Score for JSON serialization
|
||||
#[derive(Debug, Serialize)]
|
||||
pub struct ScoreJson {
|
||||
impl FffFileItem {
|
||||
/// ## Safety
|
||||
/// All string pointers must have been allocated by `CString::into_raw`.
|
||||
pub unsafe fn free_strings(&mut self) {
|
||||
unsafe {
|
||||
if !self.relative_path.is_null() {
|
||||
drop(CString::from_raw(self.relative_path));
|
||||
}
|
||||
if !self.file_name.is_null() {
|
||||
drop(CString::from_raw(self.file_name));
|
||||
}
|
||||
if !self.git_status.is_null() {
|
||||
drop(CString::from_raw(self.git_status));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Score breakdown for a search result.
|
||||
#[repr(C)]
|
||||
pub struct FffScore {
|
||||
pub total: i32,
|
||||
pub base_score: i32,
|
||||
pub filename_bonus: i32,
|
||||
@@ -132,13 +120,14 @@ pub struct ScoreJson {
|
||||
pub distance_penalty: i32,
|
||||
pub current_file_penalty: i32,
|
||||
pub combo_match_boost: i32,
|
||||
pub path_alignment_bonus: i32,
|
||||
pub exact_match: bool,
|
||||
pub match_type: String,
|
||||
pub match_type: *mut c_char,
|
||||
}
|
||||
|
||||
impl ScoreJson {
|
||||
pub fn from_score(score: &Score) -> Self {
|
||||
ScoreJson {
|
||||
impl From<&Score> for FffScore {
|
||||
fn from(score: &Score) -> Self {
|
||||
FffScore {
|
||||
total: score.total,
|
||||
base_score: score.base_score,
|
||||
filename_bonus: score.filename_bonus,
|
||||
@@ -147,77 +136,592 @@ impl ScoreJson {
|
||||
distance_penalty: score.distance_penalty,
|
||||
current_file_penalty: score.current_file_penalty,
|
||||
combo_match_boost: score.combo_match_boost,
|
||||
path_alignment_bonus: score.path_alignment_bonus,
|
||||
exact_match: score.exact_match,
|
||||
match_type: score.match_type.to_string(),
|
||||
match_type: cstring_new(score.match_type),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Location for JSON serialization
|
||||
#[derive(Debug, Serialize)]
|
||||
#[serde(tag = "type")]
|
||||
pub enum LocationJson {
|
||||
#[serde(rename = "line")]
|
||||
Line { line: i32 },
|
||||
#[serde(rename = "position")]
|
||||
Position { line: i32, col: i32 },
|
||||
#[serde(rename = "range")]
|
||||
Range {
|
||||
start: PositionJson,
|
||||
end: PositionJson,
|
||||
},
|
||||
impl FffScore {
|
||||
/// ## Safety
|
||||
/// `match_type` must have been allocated by `CString::into_raw`.
|
||||
pub unsafe fn free_strings(&mut self) {
|
||||
unsafe {
|
||||
if !self.match_type.is_null() {
|
||||
drop(CString::from_raw(self.match_type));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug, Serialize)]
|
||||
pub struct PositionJson {
|
||||
/// Location parsed from a query string (e.g. `"file.ts:42:10"`).
|
||||
///
|
||||
/// `tag` encodes the variant:
|
||||
/// 0 = no location,
|
||||
/// 1 = line only (`line` is set),
|
||||
/// 2 = position (`line` + `col`),
|
||||
/// 3 = range (`line`/`col` = start, `end_line`/`end_col` = end).
|
||||
#[repr(C)]
|
||||
pub struct FffLocation {
|
||||
pub tag: u8,
|
||||
pub line: i32,
|
||||
pub col: i32,
|
||||
pub end_line: i32,
|
||||
pub end_col: i32,
|
||||
}
|
||||
|
||||
impl LocationJson {
|
||||
pub fn from_location(loc: &Location) -> Self {
|
||||
impl From<Option<&Location>> for FffLocation {
|
||||
fn from(loc: Option<&Location>) -> Self {
|
||||
match loc {
|
||||
Location::Line(line) => LocationJson::Line { line: *line },
|
||||
Location::Position { line, col } => LocationJson::Position {
|
||||
None => FffLocation {
|
||||
tag: 0,
|
||||
line: 0,
|
||||
col: 0,
|
||||
end_line: 0,
|
||||
end_col: 0,
|
||||
},
|
||||
Some(Location::Line(line)) => FffLocation {
|
||||
tag: 1,
|
||||
line: *line,
|
||||
col: 0,
|
||||
end_line: 0,
|
||||
end_col: 0,
|
||||
},
|
||||
Some(Location::Position { line, col }) => FffLocation {
|
||||
tag: 2,
|
||||
line: *line,
|
||||
col: *col,
|
||||
end_line: 0,
|
||||
end_col: 0,
|
||||
},
|
||||
Location::Range { start, end } => LocationJson::Range {
|
||||
start: PositionJson {
|
||||
line: start.0,
|
||||
col: start.1,
|
||||
},
|
||||
end: PositionJson {
|
||||
line: end.0,
|
||||
col: end.1,
|
||||
},
|
||||
Some(Location::Range { start, end }) => FffLocation {
|
||||
tag: 3,
|
||||
line: start.0,
|
||||
col: start.1,
|
||||
end_line: end.0,
|
||||
end_col: end.1,
|
||||
},
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Search result for JSON serialization
|
||||
#[derive(Debug, Serialize)]
|
||||
pub struct SearchResultJson {
|
||||
pub items: Vec<FileItemJson>,
|
||||
pub scores: Vec<ScoreJson>,
|
||||
pub total_matched: usize,
|
||||
pub total_files: usize,
|
||||
pub location: Option<LocationJson>,
|
||||
/// Search result returned by `fff_search`.
|
||||
///
|
||||
/// The caller must free this with `fff_free_search_result`.
|
||||
#[repr(C)]
|
||||
pub struct FffSearchResult {
|
||||
/// Pointer to a heap-allocated array of `FffFileItem` (length = `count`).
|
||||
pub items: *mut FffFileItem,
|
||||
/// Pointer to a heap-allocated array of `FffScore` (length = `count`).
|
||||
pub scores: *mut FffScore,
|
||||
/// Number of items/scores in the arrays.
|
||||
pub count: u32,
|
||||
/// Total number of files that matched the query.
|
||||
pub total_matched: u32,
|
||||
/// Total number of indexed files.
|
||||
pub total_files: u32,
|
||||
/// Location parsed from the query string.
|
||||
pub location: FffLocation,
|
||||
}
|
||||
|
||||
impl SearchResultJson {
|
||||
pub fn from_search_result(result: &SearchResult) -> Self {
|
||||
SearchResultJson {
|
||||
items: result
|
||||
.items
|
||||
.iter()
|
||||
.map(|item| FileItemJson::from_file_item(item))
|
||||
.collect(),
|
||||
scores: result.scores.iter().map(ScoreJson::from_score).collect(),
|
||||
total_matched: result.total_matched,
|
||||
total_files: result.total_files,
|
||||
location: result.location.as_ref().map(LocationJson::from_location),
|
||||
impl FffSearchResult {
|
||||
/// Convert a core `SearchResult` into a heap-allocated `FffSearchResult`.
|
||||
pub fn from_core(result: &SearchResult, picker: &FilePicker) -> *mut Self {
|
||||
let items: Vec<FffFileItem> = result
|
||||
.items
|
||||
.iter()
|
||||
.map(|i| FffFileItem::from_item(i, picker))
|
||||
.collect();
|
||||
let scores: Vec<FffScore> = result.scores.iter().map(FffScore::from).collect();
|
||||
let count = items.len() as u32;
|
||||
|
||||
let (items_ptr, _) = vec_to_raw(items);
|
||||
let (scores_ptr, _) = vec_to_raw(scores);
|
||||
|
||||
Box::into_raw(Box::new(FffSearchResult {
|
||||
items: items_ptr,
|
||||
scores: scores_ptr,
|
||||
count,
|
||||
total_matched: result.total_matched as u32,
|
||||
total_files: result.total_files as u32,
|
||||
location: FffLocation::from(result.location.as_ref()),
|
||||
}))
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Grep result types
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// A byte range within a matched line, used for highlighting.
|
||||
#[repr(C)]
|
||||
pub struct FffMatchRange {
|
||||
pub start: u32,
|
||||
pub end: u32,
|
||||
}
|
||||
|
||||
/// A single grep match with file and line information.
|
||||
///
|
||||
/// All string fields and arrays are heap-allocated. Free the parent
|
||||
/// `FffGrepResult` with `fff_free_grep_result` to release everything.
|
||||
#[repr(C)]
|
||||
pub struct FffGrepMatch {
|
||||
// -- pointers (8 bytes each) --
|
||||
pub relative_path: *mut c_char,
|
||||
pub file_name: *mut c_char,
|
||||
pub git_status: *mut c_char,
|
||||
pub line_content: *mut c_char,
|
||||
pub match_ranges: *mut FffMatchRange,
|
||||
pub context_before: *mut *mut c_char,
|
||||
pub context_after: *mut *mut c_char,
|
||||
// -- 8-byte numeric fields --
|
||||
pub size: u64,
|
||||
pub modified: u64,
|
||||
pub total_frecency_score: i64,
|
||||
pub access_frecency_score: i64,
|
||||
pub modification_frecency_score: i64,
|
||||
pub line_number: u64,
|
||||
pub byte_offset: u64,
|
||||
// -- 4-byte fields --
|
||||
pub col: u32,
|
||||
pub match_ranges_count: u32,
|
||||
pub context_before_count: u32,
|
||||
pub context_after_count: u32,
|
||||
// -- 2-byte fields --
|
||||
pub fuzzy_score: u16,
|
||||
// -- 1-byte fields --
|
||||
pub has_fuzzy_score: bool,
|
||||
pub is_binary: bool,
|
||||
pub is_definition: bool,
|
||||
}
|
||||
|
||||
impl FffGrepMatch {
|
||||
fn from_core_with_file(m: &GrepMatch, file: &FileItem, picker: &FilePicker) -> Self {
|
||||
let ranges: Vec<FffMatchRange> = m
|
||||
.match_byte_offsets
|
||||
.iter()
|
||||
.map(|&(start, end)| FffMatchRange { start, end })
|
||||
.collect();
|
||||
let (match_ranges, match_ranges_count) = vec_to_raw(ranges);
|
||||
let (context_before, context_before_count) = strings_to_raw(&m.context_before);
|
||||
let (context_after, context_after_count) = strings_to_raw(&m.context_after);
|
||||
let (has_fuzzy_score, fuzzy_score) = match m.fuzzy_score {
|
||||
Some(s) => (true, s),
|
||||
None => (false, 0),
|
||||
};
|
||||
|
||||
FffGrepMatch {
|
||||
relative_path: cstring_new(&file.relative_path(picker)),
|
||||
file_name: cstring_new(&file.file_name(picker)),
|
||||
git_status: cstring_new(format_git_status(file.git_status)),
|
||||
line_content: cstring_new(&m.line_content),
|
||||
match_ranges,
|
||||
context_before,
|
||||
context_after,
|
||||
size: file.size,
|
||||
modified: file.modified,
|
||||
total_frecency_score: file.total_frecency_score() as i64,
|
||||
access_frecency_score: file.access_frecency_score as i64,
|
||||
modification_frecency_score: file.modification_frecency_score as i64,
|
||||
line_number: m.line_number,
|
||||
byte_offset: m.byte_offset,
|
||||
col: m.col as u32,
|
||||
match_ranges_count,
|
||||
context_before_count,
|
||||
context_after_count,
|
||||
fuzzy_score,
|
||||
has_fuzzy_score,
|
||||
is_binary: file.is_binary(),
|
||||
is_definition: m.is_definition,
|
||||
}
|
||||
}
|
||||
|
||||
/// ## Safety
|
||||
/// All pointers must have been allocated by the corresponding `from_core`.
|
||||
pub unsafe fn free_fields(&mut self) {
|
||||
unsafe {
|
||||
if !self.relative_path.is_null() {
|
||||
drop(CString::from_raw(self.relative_path));
|
||||
}
|
||||
if !self.file_name.is_null() {
|
||||
drop(CString::from_raw(self.file_name));
|
||||
}
|
||||
if !self.git_status.is_null() {
|
||||
drop(CString::from_raw(self.git_status));
|
||||
}
|
||||
if !self.line_content.is_null() {
|
||||
drop(CString::from_raw(self.line_content));
|
||||
}
|
||||
if !self.match_ranges.is_null() {
|
||||
drop(Vec::from_raw_parts(
|
||||
self.match_ranges,
|
||||
self.match_ranges_count as usize,
|
||||
self.match_ranges_count as usize,
|
||||
));
|
||||
}
|
||||
free_cstring_array(self.context_before, self.context_before_count);
|
||||
free_cstring_array(self.context_after, self.context_after_count);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Grep result returned by `fff_live_grep` and `fff_multi_grep`.
|
||||
///
|
||||
/// The caller must free this with `fff_free_grep_result`.
|
||||
#[repr(C)]
|
||||
pub struct FffGrepResult {
|
||||
/// Pointer to a heap-allocated array of `FffGrepMatch` (length = `count`).
|
||||
pub items: *mut FffGrepMatch,
|
||||
/// Number of matches in the `items` array.
|
||||
pub count: u32,
|
||||
/// Total number of matches (always equal to `count`).
|
||||
pub total_matched: u32,
|
||||
/// Number of files actually opened and searched in this call.
|
||||
pub total_files_searched: u32,
|
||||
/// Total number of indexed files (before any filtering).
|
||||
pub total_files: u32,
|
||||
/// Number of files eligible for search after filtering.
|
||||
pub filtered_file_count: u32,
|
||||
/// File offset for the next page. 0 if all files have been searched.
|
||||
pub next_file_offset: u32,
|
||||
/// Regex compilation error when falling back to literal matching. Null if none.
|
||||
pub regex_fallback_error: *mut c_char,
|
||||
}
|
||||
|
||||
impl FffGrepResult {
|
||||
/// Convert a core `GrepResult` into a heap-allocated `FffGrepResult`.
|
||||
pub fn from_core(result: &GrepResult, picker: &FilePicker) -> *mut Self {
|
||||
let items: Vec<FffGrepMatch> = result
|
||||
.matches
|
||||
.iter()
|
||||
.map(|m| {
|
||||
let file = result.files[m.file_index];
|
||||
FffGrepMatch::from_core_with_file(m, file, picker)
|
||||
})
|
||||
.collect();
|
||||
let (items_ptr, count) = vec_to_raw(items);
|
||||
|
||||
Box::into_raw(Box::new(FffGrepResult {
|
||||
items: items_ptr,
|
||||
count,
|
||||
total_matched: result.matches.len() as u32,
|
||||
total_files_searched: result.total_files_searched as u32,
|
||||
total_files: result.total_files as u32,
|
||||
filtered_file_count: result.filtered_file_count as u32,
|
||||
next_file_offset: result.next_file_offset as u32,
|
||||
regex_fallback_error: match &result.regex_fallback_error {
|
||||
Some(e) => cstring_new(e),
|
||||
None => ptr::null_mut(),
|
||||
},
|
||||
}))
|
||||
}
|
||||
}
|
||||
|
||||
/// Result envelope returned by all `fff_*` functions.
|
||||
///
|
||||
/// Heap-allocated — the caller must free it with `fff_free_result`.
|
||||
///
|
||||
/// Depending on the function, the payload is delivered through different fields:
|
||||
///
|
||||
/// | Function | Payload field | Type |
|
||||
/// |----------------------------|---------------|-------------------------------|
|
||||
/// | `fff_create_instance` | `handle` | opaque instance pointer |
|
||||
/// | `fff_search` | `handle` | `*mut FffSearchResult` |
|
||||
/// | `fff_live_grep` | `handle` | `*mut FffGrepResult` |
|
||||
/// | `fff_multi_grep` | `handle` | `*mut FffGrepResult` |
|
||||
/// | `fff_get_scan_progress` | `handle` | `*mut FffScanProgress` |
|
||||
/// | `fff_health_check` | `handle` | `*mut c_char` (JSON string) |
|
||||
/// | `fff_get_historical_query` | `handle` | `*mut c_char` (string or null)|
|
||||
/// | `fff_wait_for_scan` | `int_value` | 1 = completed, 0 = timed out |
|
||||
/// | `fff_track_query` | `int_value` | 1 = success, 0 = failure |
|
||||
/// | `fff_refresh_git_status` | `int_value` | number of files updated |
|
||||
/// | `fff_scan_files` | (none) | success flag only |
|
||||
/// | `fff_restart_index` | (none) | success flag only |
|
||||
///
|
||||
/// On failure, `success` is false and `error` contains the message.
|
||||
///
|
||||
/// **Important:** `fff_free_result` frees `error` but does **not** free `handle`.
|
||||
/// The caller must free the handle with the appropriate function
|
||||
/// (`fff_destroy`, `fff_free_search_result`, `fff_free_grep_result`,
|
||||
/// `fff_free_string`, etc.).
|
||||
#[repr(C)]
|
||||
pub struct FffResult {
|
||||
/// Whether the operation succeeded.
|
||||
pub success: bool,
|
||||
/// Error message on failure. Null on success.
|
||||
pub error: *mut c_char,
|
||||
/// Opaque pointer payload (instance handle, typed result struct, or string). May be null.
|
||||
pub handle: *mut c_void,
|
||||
/// Integer payload for simple return values (bool as 0/1, counts, etc.).
|
||||
pub int_value: i64,
|
||||
}
|
||||
|
||||
impl FffResult {
|
||||
/// Create a successful result with no payload, returned as heap pointer.
|
||||
pub fn ok_empty() -> *mut Self {
|
||||
Box::into_raw(Box::new(FffResult {
|
||||
success: true,
|
||||
error: ptr::null_mut(),
|
||||
handle: ptr::null_mut(),
|
||||
int_value: 0,
|
||||
}))
|
||||
}
|
||||
|
||||
/// Create a successful result with an integer value.
|
||||
pub fn ok_int(value: i64) -> *mut Self {
|
||||
Box::into_raw(Box::new(FffResult {
|
||||
success: true,
|
||||
error: ptr::null_mut(),
|
||||
handle: ptr::null_mut(),
|
||||
int_value: value,
|
||||
}))
|
||||
}
|
||||
|
||||
/// Create a successful result carrying an opaque pointer (handle, typed struct, or string).
|
||||
pub fn ok_handle(handle: *mut c_void) -> *mut Self {
|
||||
Box::into_raw(Box::new(FffResult {
|
||||
success: true,
|
||||
error: ptr::null_mut(),
|
||||
handle,
|
||||
int_value: 0,
|
||||
}))
|
||||
}
|
||||
|
||||
/// Create a successful result carrying a C string in the `handle` field.
|
||||
/// The caller must free it with `fff_free_string`.
|
||||
pub fn ok_string(s: &str) -> *mut Self {
|
||||
let cstr = CString::new(s).unwrap_or_default().into_raw();
|
||||
Box::into_raw(Box::new(FffResult {
|
||||
success: true,
|
||||
error: ptr::null_mut(),
|
||||
handle: cstr as *mut c_void,
|
||||
int_value: 0,
|
||||
}))
|
||||
}
|
||||
|
||||
/// Create an error result, returned as heap pointer.
|
||||
pub fn err(error: &str) -> *mut Self {
|
||||
Box::into_raw(Box::new(FffResult {
|
||||
success: false,
|
||||
error: CString::new(error).unwrap_or_default().into_raw(),
|
||||
handle: ptr::null_mut(),
|
||||
int_value: 0,
|
||||
}))
|
||||
}
|
||||
}
|
||||
|
||||
/// A directory item returned by `fff_search_directories`.
|
||||
///
|
||||
/// All string fields are heap-allocated and owned by the parent `FffDirSearchResult`.
|
||||
/// Free the entire result with `fff_free_dir_search_result`.
|
||||
#[repr(C)]
|
||||
pub struct FffDirItem {
|
||||
pub relative_path: *mut c_char,
|
||||
pub dir_name: *mut c_char,
|
||||
pub max_access_frecency: i32,
|
||||
}
|
||||
|
||||
impl FffDirItem {
|
||||
pub fn from_item(item: &DirItem, picker: &FilePicker) -> Self {
|
||||
FffDirItem {
|
||||
relative_path: cstring_new(&item.relative_path(picker)),
|
||||
dir_name: cstring_new(&item.dir_name(picker)),
|
||||
max_access_frecency: item.max_access_frecency(),
|
||||
}
|
||||
}
|
||||
|
||||
/// ## Safety
|
||||
/// All string pointers must have been allocated by the rust side
|
||||
pub unsafe fn free_strings(&mut self) {
|
||||
unsafe {
|
||||
if !self.relative_path.is_null() {
|
||||
drop(CString::from_raw(self.relative_path));
|
||||
}
|
||||
if !self.dir_name.is_null() {
|
||||
drop(CString::from_raw(self.dir_name));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Directory search result returned by `fff_search_directories`.
|
||||
///
|
||||
/// The caller must free this with `fff_free_dir_search_result`.
|
||||
#[repr(C)]
|
||||
pub struct FffDirSearchResult {
|
||||
/// Pointer to a heap-allocated array of `FffDirItem` (length = `count`).
|
||||
pub items: *mut FffDirItem,
|
||||
/// Pointer to a heap-allocated array of `FffScore` (length = `count`).
|
||||
pub scores: *mut FffScore,
|
||||
/// Number of items/scores in the arrays.
|
||||
pub count: u32,
|
||||
/// Total number of directories that matched the query.
|
||||
pub total_matched: u32,
|
||||
/// Total number of indexed directories.
|
||||
pub total_dirs: u32,
|
||||
}
|
||||
|
||||
impl FffDirSearchResult {
|
||||
/// Convert a core `DirSearchResult` into a heap-allocated `FffDirSearchResult`.
|
||||
pub fn from_core(result: &DirSearchResult, picker: &FilePicker) -> *mut Self {
|
||||
let items: Vec<FffDirItem> = result
|
||||
.items
|
||||
.iter()
|
||||
.map(|i| FffDirItem::from_item(i, picker))
|
||||
.collect();
|
||||
let scores: Vec<FffScore> = result.scores.iter().map(FffScore::from).collect();
|
||||
let count = items.len() as u32;
|
||||
|
||||
let (items_ptr, _) = vec_to_raw(items);
|
||||
let (scores_ptr, _) = vec_to_raw(scores);
|
||||
|
||||
Box::into_raw(Box::new(FffDirSearchResult {
|
||||
items: items_ptr,
|
||||
scores: scores_ptr,
|
||||
count,
|
||||
total_matched: result.total_matched as u32,
|
||||
total_dirs: result.total_dirs as u32,
|
||||
}))
|
||||
}
|
||||
}
|
||||
|
||||
/// A single item in a mixed (files + directories) search result.
|
||||
///
|
||||
/// `item_type`: 0 = file, 1 = directory.
|
||||
/// All string fields are heap-allocated and owned by the parent `FffMixedSearchResult`.
|
||||
#[repr(C)]
|
||||
pub struct FffMixedItem {
|
||||
/// 0 = file, 1 = directory.
|
||||
pub item_type: u8,
|
||||
pub relative_path: *mut c_char,
|
||||
/// Filename for files, last directory segment for directories.
|
||||
pub display_name: *mut c_char,
|
||||
pub git_status: *mut c_char,
|
||||
pub size: u64,
|
||||
pub modified: u64,
|
||||
/// The access frecency score for files, or max access frecency among all the immediate
|
||||
/// children for directories.
|
||||
pub access_frecency_score: i64,
|
||||
/// Always 0 for directories
|
||||
pub modification_frecency_score: i64,
|
||||
/// Always 0 for directories
|
||||
pub total_frecency_score: i64,
|
||||
/// Always 0 for directories
|
||||
pub is_binary: bool,
|
||||
}
|
||||
|
||||
impl FffMixedItem {
|
||||
pub fn from_mixed_ref(item: &MixedItemRef<'_>, picker: &FilePicker) -> Self {
|
||||
match item {
|
||||
MixedItemRef::File(file) => FffMixedItem {
|
||||
item_type: 0,
|
||||
relative_path: cstring_new(&file.relative_path(picker)),
|
||||
display_name: cstring_new(&file.file_name(picker)),
|
||||
git_status: cstring_new(format_git_status(file.git_status)),
|
||||
size: file.size,
|
||||
modified: file.modified,
|
||||
access_frecency_score: file.access_frecency_score as i64,
|
||||
modification_frecency_score: file.modification_frecency_score as i64,
|
||||
total_frecency_score: file.total_frecency_score() as i64,
|
||||
is_binary: file.is_binary(),
|
||||
},
|
||||
MixedItemRef::Dir(dir) => FffMixedItem {
|
||||
item_type: 1,
|
||||
relative_path: cstring_new(&dir.relative_path(picker)),
|
||||
display_name: cstring_new(&dir.dir_name(picker)),
|
||||
git_status: cstring_new(""),
|
||||
size: 0,
|
||||
modified: 0,
|
||||
access_frecency_score: dir.max_access_frecency() as i64,
|
||||
modification_frecency_score: 0,
|
||||
total_frecency_score: dir.max_access_frecency() as i64,
|
||||
is_binary: false,
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
/// ## Safety
|
||||
/// All string pointers must have been allocated by rust side
|
||||
pub unsafe fn free_strings(&mut self) {
|
||||
unsafe {
|
||||
if !self.relative_path.is_null() {
|
||||
drop(CString::from_raw(self.relative_path));
|
||||
}
|
||||
if !self.display_name.is_null() {
|
||||
drop(CString::from_raw(self.display_name));
|
||||
}
|
||||
if !self.git_status.is_null() {
|
||||
drop(CString::from_raw(self.git_status));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Mixed search result returned by `fff_search_mixed`.
|
||||
///
|
||||
/// The caller must free this with `fff_free_mixed_search_result`.
|
||||
#[repr(C)]
|
||||
pub struct FffMixedSearchResult {
|
||||
/// Pointer to a heap-allocated array of `FffMixedItem` (length = `count`).
|
||||
pub items: *mut FffMixedItem,
|
||||
/// Pointer to a heap-allocated array of `FffScore` (length = `count`).
|
||||
pub scores: *mut FffScore,
|
||||
/// Number of items/scores in the arrays.
|
||||
pub count: u32,
|
||||
/// Total number of items (files + dirs) that matched the query.
|
||||
pub total_matched: u32,
|
||||
/// Total number of indexed files.
|
||||
pub total_files: u32,
|
||||
/// Total number of indexed directories.
|
||||
pub total_dirs: u32,
|
||||
/// Location parsed from the query string.
|
||||
pub location: FffLocation,
|
||||
}
|
||||
|
||||
impl FffMixedSearchResult {
|
||||
/// Convert a core `MixedSearchResult` into a heap-allocated `FffMixedSearchResult`.
|
||||
pub fn from_core(result: &MixedSearchResult, picker: &FilePicker) -> *mut Self {
|
||||
let items: Vec<FffMixedItem> = result
|
||||
.items
|
||||
.iter()
|
||||
.map(|i| FffMixedItem::from_mixed_ref(i, picker))
|
||||
.collect();
|
||||
let scores: Vec<FffScore> = result.scores.iter().map(FffScore::from).collect();
|
||||
let count = items.len() as u32;
|
||||
|
||||
let (items_ptr, _) = vec_to_raw(items);
|
||||
let (scores_ptr, _) = vec_to_raw(scores);
|
||||
|
||||
Box::into_raw(Box::new(FffMixedSearchResult {
|
||||
items: items_ptr,
|
||||
scores: scores_ptr,
|
||||
count,
|
||||
total_matched: result.total_matched as u32,
|
||||
total_files: result.total_files as u32,
|
||||
total_dirs: result.total_dirs as u32,
|
||||
location: FffLocation::from(result.location.as_ref()),
|
||||
}))
|
||||
}
|
||||
}
|
||||
|
||||
/// Scan progress returned by `fff_get_scan_progress`.
|
||||
/// The caller must free this with `fff_free_scan_progress`.
|
||||
#[repr(C)]
|
||||
pub struct FffScanProgress {
|
||||
pub scanned_files_count: u64,
|
||||
pub is_scanning: bool,
|
||||
pub is_watcher_ready: bool,
|
||||
pub is_warmup_complete: bool,
|
||||
}
|
||||
|
||||
impl From<fff::file_picker::ScanProgress> for FffScanProgress {
|
||||
fn from(p: fff::file_picker::ScanProgress) -> Self {
|
||||
Self {
|
||||
scanned_files_count: p.scanned_files_count as u64,
|
||||
is_scanning: p.is_scanning,
|
||||
is_watcher_ready: p.is_watcher_ready,
|
||||
is_warmup_complete: p.is_warmup_complete,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+1271
-408
File diff suppressed because it is too large
Load Diff
+33
-21
@@ -1,39 +1,53 @@
|
||||
[package]
|
||||
name = "fff-core"
|
||||
version = "0.1.0"
|
||||
name = "fff-search"
|
||||
version = "0.8.1"
|
||||
edition = "2024"
|
||||
description = "High-performance file finder core library"
|
||||
license = "MIT"
|
||||
authors = ["Dmitriy Kovalenko <dmtr.kovalenko@outlook.com>"]
|
||||
description = "Faboulous & Fast File Finder - a fast and extremely correct file finder SDK with typo resistance, SIMD, prefiltering, and more"
|
||||
|
||||
[lib]
|
||||
path = "src/lib.rs"
|
||||
crate-type = ["rlib", "staticlib", "cdylib"]
|
||||
|
||||
[[bench]]
|
||||
name = "parse_bench"
|
||||
harness = false
|
||||
|
||||
[[bench]]
|
||||
name = "bigram_bench"
|
||||
harness = false
|
||||
|
||||
[[bench]]
|
||||
name = "memmem_bench"
|
||||
harness = false
|
||||
|
||||
[features]
|
||||
default = []
|
||||
# Enable C FFI exports
|
||||
ffi = []
|
||||
# Call mi_collect(true) after large allocator churn (bigram build).
|
||||
# Requires mimalloc to be the global allocator (linked by fff-nvim).
|
||||
mimalloc-collect = ["dep:libmimalloc-sys"]
|
||||
# Use zlob (Zig-compiled C globbing library) for glob matching.
|
||||
# Requires Zig to be installed. When disabled, falls back to globset (pure Rust).
|
||||
zlob = ["dep:zlob", "fff-query-parser/zlob"]
|
||||
|
||||
[dependencies]
|
||||
# Workspace dependencies
|
||||
ahash = { workspace = true }
|
||||
rayon = { workspace = true }
|
||||
smallvec = { workspace = true }
|
||||
thiserror = { workspace = true }
|
||||
tracing = { workspace = true }
|
||||
|
||||
# Local crates
|
||||
fff-query-parser = { path = "../fff-query-parser" }
|
||||
|
||||
# External dependencies
|
||||
bindet = { workspace = true }
|
||||
fff-query-parser = { workspace = true }
|
||||
blake3 = { workspace = true }
|
||||
chrono = { workspace = true }
|
||||
dirs = { workspace = true }
|
||||
libc = "0.2"
|
||||
git2 = { workspace = true }
|
||||
glidesort = { workspace = true }
|
||||
grep-matcher = { workspace = true }
|
||||
grep-searcher = { workspace = true }
|
||||
globset = { workspace = true }
|
||||
fff-grep = { workspace = true }
|
||||
aho-corasick = "1"
|
||||
memchr = "2"
|
||||
heed = { workspace = true }
|
||||
ignore = { workspace = true }
|
||||
@@ -41,25 +55,23 @@ memmap2 = { workspace = true }
|
||||
neo_frizbee = { workspace = true }
|
||||
notify = { workspace = true }
|
||||
notify-debouncer-full = { workspace = true }
|
||||
once_cell = { workspace = true }
|
||||
parking_lot = { workspace = true }
|
||||
pathdiff = { workspace = true }
|
||||
regex = { workspace = true }
|
||||
regex-syntax = "0.8"
|
||||
serde = { version = "1.0", features = ["derive"] }
|
||||
smartstring = { version = "1.0.1", features = ["serde"] }
|
||||
tracing-appender = "0.2"
|
||||
tracing-subscriber = { version = "0.3", features = ["env-filter"] }
|
||||
zlob = { workspace = true }
|
||||
|
||||
# Platform-specific: Use vendored OpenSSL on non-Windows (Linux, macOS)
|
||||
[target.'cfg(not(windows))'.dependencies]
|
||||
openssl = { version = "0.10", features = ["vendored"] }
|
||||
|
||||
zlob = { workspace = true, optional = true }
|
||||
libmimalloc-sys = { version = "0.1", optional = true, features = ["extended", "local_dynamic_tls"] }
|
||||
mimalloc = { version = "0.1", optional = true, features = ["local_dynamic_tls"] }
|
||||
# Platform-specific: dunce for Windows to avoid \\?\ extended path prefix
|
||||
[target.'cfg(windows)'.dependencies]
|
||||
dunce = { workspace = true }
|
||||
|
||||
[dev-dependencies]
|
||||
criterion = { version = "0.5", features = ["html_reports"] }
|
||||
ctor = "0.2"
|
||||
proptest = { version = "1", default-features = false, features = ["std", "fork"] }
|
||||
rand = { version = "0.8", features = ["small_rng"] }
|
||||
tempfile = "3.8"
|
||||
|
||||
@@ -0,0 +1,26 @@
|
||||
# fff
|
||||
|
||||
fff is a file search toolkit. It is faster than ripgrep and fzf and designed for a long running applications like file editors, ai agents, or file exploerers.
|
||||
|
||||
## Features
|
||||
|
||||
- Fuzzy file name search
|
||||
- Typo resistance
|
||||
- Frecency and query history ranking
|
||||
- Native git support via libgit
|
||||
- Advanced ranking
|
||||
- Grep functionality with SIMD optimized plain matcher and regex
|
||||
- Multi grep using aho-corasick algorithm
|
||||
- Efficient memory mapping for file system
|
||||
- Cross platform support (Linux, Windows, MacOS)
|
||||
- Advnaced constraints syntax allowing to prefilter based on git status, glob, extension, size, timing and more
|
||||
|
||||
## Performance
|
||||
|
||||
FFF is designed for high performance and low latency. SIMD optimized where needed, parallelized for multi core systems, efficient sorting and ranking algorithms, memaps and much more.
|
||||
|
||||
On MacOS FFF is about 20-50 times faster than ripgrep for content search and around 10 times faster than fzf for file name search.
|
||||
|
||||
## Documentation
|
||||
|
||||
Refer rust docs https://docs.rs/crate/fff-search/latest
|
||||
@@ -0,0 +1,164 @@
|
||||
use criterion::{BenchmarkId, Criterion, black_box, criterion_group, criterion_main};
|
||||
use fff_search::bigram_filter::{BigramFilter, BigramIndexBuilder};
|
||||
|
||||
/// Build a realistic bigram index for benchmarking.
|
||||
/// Simulates a large repo by generating varied content per file.
|
||||
fn build_test_index(file_count: usize) -> BigramFilter {
|
||||
let builder = BigramIndexBuilder::new(file_count);
|
||||
let skip_builder = BigramIndexBuilder::new(file_count);
|
||||
|
||||
for i in 0..file_count {
|
||||
// Generate varied content so we get a mix of sparse and dense columns
|
||||
let content = format!(
|
||||
"struct File{i} {{ fn process() {{ let controller = read(path); }} }} // module {i}"
|
||||
);
|
||||
builder.add_file_content(&skip_builder, i, content.as_bytes());
|
||||
}
|
||||
|
||||
let mut index = builder.compress(None);
|
||||
let skip_index = skip_builder.compress(Some(12));
|
||||
index.set_skip_index(skip_index);
|
||||
index
|
||||
}
|
||||
|
||||
fn bench_bigram_query(c: &mut Criterion) {
|
||||
let file_counts = [10_000, 100_000, 500_000];
|
||||
|
||||
for &file_count in &file_counts {
|
||||
let index = build_test_index(file_count);
|
||||
eprintln!(
|
||||
"Index ({} files): {} columns",
|
||||
file_count,
|
||||
index.columns_used(),
|
||||
);
|
||||
|
||||
let mut group = c.benchmark_group(format!("bigram_query_{file_count}"));
|
||||
group.sample_size(500);
|
||||
|
||||
let queries: &[(&str, &[u8])] = &[
|
||||
("short_2char", b"st"),
|
||||
("medium_6char", b"struct"),
|
||||
("long_14char", b"let controller"),
|
||||
("multi_word", b"fn process"),
|
||||
];
|
||||
|
||||
for (name, query) in queries {
|
||||
group.bench_with_input(BenchmarkId::from_parameter(name), query, |b, q| {
|
||||
b.iter(|| {
|
||||
let result = index.query(black_box(q));
|
||||
black_box(&result);
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
group.finish();
|
||||
}
|
||||
}
|
||||
|
||||
fn bench_bigram_is_candidate(c: &mut Criterion) {
|
||||
let index = build_test_index(500_000);
|
||||
let candidates = match index.query(b"struct") {
|
||||
Some(c) => c,
|
||||
None => {
|
||||
// All bigrams ubiquitous at this size — skip candidate benches
|
||||
eprintln!("Skipping is_candidate bench: query returned None (all bigrams ubiquitous)");
|
||||
return;
|
||||
}
|
||||
};
|
||||
|
||||
c.bench_function("is_candidate_500k", |b| {
|
||||
b.iter(|| {
|
||||
let mut count = 0u32;
|
||||
for i in 0..500_000 {
|
||||
if BigramFilter::is_candidate(black_box(&candidates), i) {
|
||||
count += 1;
|
||||
}
|
||||
}
|
||||
black_box(count)
|
||||
});
|
||||
});
|
||||
|
||||
c.bench_function("count_candidates_500k", |b| {
|
||||
b.iter(|| BigramFilter::count_candidates(black_box(&candidates)));
|
||||
});
|
||||
}
|
||||
|
||||
fn bench_bigram_build(c: &mut Criterion) {
|
||||
let mut group = c.benchmark_group("bigram_build");
|
||||
group.sample_size(10);
|
||||
|
||||
let file_counts = [10_000, 100_000];
|
||||
|
||||
for &file_count in &file_counts {
|
||||
// Pre-generate content so we only measure index building.
|
||||
// Short content (~85 bytes/file) exercises the scalar fast path.
|
||||
let contents: Vec<String> = (0..file_count)
|
||||
.map(|i| {
|
||||
format!(
|
||||
"struct File{i} {{ fn process() {{ let controller = read(path); }} }} // mod {i}"
|
||||
)
|
||||
})
|
||||
.collect();
|
||||
|
||||
group.bench_with_input(
|
||||
BenchmarkId::new("short_content", file_count),
|
||||
&file_count,
|
||||
|b, &fc| {
|
||||
b.iter(|| {
|
||||
let builder = BigramIndexBuilder::new(fc);
|
||||
let skip_builder = BigramIndexBuilder::new(fc);
|
||||
for (i, content) in contents.iter().enumerate() {
|
||||
builder.add_file_content(&skip_builder, i, content.as_bytes());
|
||||
}
|
||||
let index = builder.compress(None);
|
||||
black_box(index.columns_used())
|
||||
});
|
||||
},
|
||||
);
|
||||
|
||||
// Long content (~4 KB/file) exercises the SIMD pre-pass path.
|
||||
// Build a realistic-looking source-like blob by repeating snippets.
|
||||
let long_contents: Vec<String> = (0..file_count)
|
||||
.map(|i| {
|
||||
let mut s = String::with_capacity(4096);
|
||||
for j in 0..50 {
|
||||
s.push_str(&format!(
|
||||
"pub fn handler_{i}_{j}(ctx: &Context) -> Result<Response, Error> {{\n"
|
||||
));
|
||||
s.push_str(" let parsed = ctx.parse()?;\n");
|
||||
s.push_str(" let validated = parsed.validate()?;\n");
|
||||
s.push_str(&format!(" ctx.respond(validated, {}).await\n", j));
|
||||
s.push_str("}\n\n");
|
||||
}
|
||||
s
|
||||
})
|
||||
.collect();
|
||||
|
||||
group.bench_with_input(
|
||||
BenchmarkId::new("long_content", file_count),
|
||||
&file_count,
|
||||
|b, &fc| {
|
||||
b.iter(|| {
|
||||
let builder = BigramIndexBuilder::new(fc);
|
||||
let skip_builder = BigramIndexBuilder::new(fc);
|
||||
for (i, content) in long_contents.iter().enumerate() {
|
||||
builder.add_file_content(&skip_builder, i, content.as_bytes());
|
||||
}
|
||||
let index = builder.compress(None);
|
||||
black_box(index.columns_used())
|
||||
});
|
||||
},
|
||||
);
|
||||
}
|
||||
|
||||
group.finish();
|
||||
}
|
||||
|
||||
criterion_group!(
|
||||
benches,
|
||||
bench_bigram_query,
|
||||
bench_bigram_is_candidate,
|
||||
bench_bigram_build,
|
||||
);
|
||||
|
||||
criterion_main!(benches);
|
||||
@@ -0,0 +1,93 @@
|
||||
use criterion::{BenchmarkId, Criterion, black_box, criterion_group, criterion_main};
|
||||
use fff_search::case_insensitive_memmem;
|
||||
use std::path::Path;
|
||||
|
||||
/// Load real source files from the repository as benchmark haystacks.
|
||||
/// Falls back to concatenating all .rs files under crates/ if specific files are missing.
|
||||
fn load_real_files() -> Vec<(&'static str, Vec<u8>)> {
|
||||
let manifest_dir = env!("CARGO_MANIFEST_DIR"); // crates/fff-core
|
||||
let repo_root = Path::new(manifest_dir).parent().unwrap().parent().unwrap();
|
||||
|
||||
let files: &[(&str, &str)] = &[
|
||||
("grep.rs/80KB", "crates/fff-core/src/grep.rs"),
|
||||
("file_picker.rs/53KB", "crates/fff-core/src/file_picker.rs"),
|
||||
("picker_ui.lua/96KB", "lua/fff/picker_ui.lua"),
|
||||
];
|
||||
|
||||
let mut result = Vec::new();
|
||||
for &(label, rel_path) in files {
|
||||
let full_path = repo_root.join(rel_path);
|
||||
if let Ok(data) = std::fs::read(&full_path) {
|
||||
result.push((label, data));
|
||||
}
|
||||
}
|
||||
|
||||
// Also create a large synthetic file by concatenating all three
|
||||
if result.len() == 3 {
|
||||
let mut combined = Vec::new();
|
||||
for (_, data) in &result {
|
||||
combined.extend_from_slice(data);
|
||||
}
|
||||
// Repeat to get ~1MB
|
||||
let base = combined.clone();
|
||||
while combined.len() < 1024 * 1024 {
|
||||
combined.extend_from_slice(&base);
|
||||
}
|
||||
combined.truncate(1024 * 1024);
|
||||
result.push(("combined/1MB", combined));
|
||||
}
|
||||
|
||||
result
|
||||
}
|
||||
|
||||
fn bench_memmem(c: &mut Criterion) {
|
||||
let mut group = c.benchmark_group("case_insensitive_memmem");
|
||||
|
||||
let files = load_real_files();
|
||||
assert!(!files.is_empty(), "No source files found for benchmarking");
|
||||
|
||||
// Needles chosen to exercise different false-positive rates:
|
||||
//
|
||||
// "hit" needles: strings that actually appear in these source files.
|
||||
// "miss" needles: strings with common first-bytes (lots of false positives
|
||||
// for memchr2) but that don't exist in any of the files.
|
||||
let needles: &[(&str, &[u8])] = &[
|
||||
// Hits — real identifiers from the codebase
|
||||
("short/hit/fn", b"fn"),
|
||||
("short/hit/self", b"self"),
|
||||
("medium/hit", b"search_file"),
|
||||
("long/hit", b"content_cache_budget"),
|
||||
// Misses — common first-bytes, guaranteed not in source
|
||||
("short/miss", b"zqxjv"),
|
||||
("medium/miss", b"fluxcapacitor"),
|
||||
("long/miss", b"quantum_entanglement_resolver"),
|
||||
];
|
||||
|
||||
for (file_label, haystack) in &files {
|
||||
for &(needle_label, needle) in needles {
|
||||
let needle_lower: Vec<u8> = needle.iter().map(|b| b.to_ascii_lowercase()).collect();
|
||||
let id = format!("{file_label}/{needle_label}");
|
||||
|
||||
group.bench_with_input(
|
||||
BenchmarkId::new("packed_pair", &id),
|
||||
&(haystack, &needle_lower),
|
||||
|b, &(h, n)| {
|
||||
b.iter(|| black_box(case_insensitive_memmem::search_packed_pair(h, n)));
|
||||
},
|
||||
);
|
||||
|
||||
group.bench_with_input(
|
||||
BenchmarkId::new("memchr2_search", &id),
|
||||
&(haystack, &needle_lower),
|
||||
|b, &(h, n)| {
|
||||
b.iter(|| black_box(case_insensitive_memmem::search(h, n)));
|
||||
},
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
group.finish();
|
||||
}
|
||||
|
||||
criterion_group!(benches, bench_memmem);
|
||||
criterion_main!(benches);
|
||||
@@ -0,0 +1,180 @@
|
||||
use criterion::{BenchmarkId, Criterion, Throughput, black_box, criterion_group, criterion_main};
|
||||
use fff_query_parser::*;
|
||||
|
||||
fn bench_parse_simple(c: &mut Criterion) {
|
||||
let parser = QueryParser::default();
|
||||
|
||||
c.bench_function("parse_simple_text", |b| {
|
||||
b.iter(|| parser.parse(black_box("hello world")));
|
||||
});
|
||||
|
||||
c.bench_function("parse_extension", |b| {
|
||||
b.iter(|| parser.parse(black_box("*.rs")));
|
||||
});
|
||||
|
||||
c.bench_function("parse_text_with_extension", |b| {
|
||||
b.iter(|| parser.parse(black_box("name *.rs")));
|
||||
});
|
||||
}
|
||||
|
||||
fn bench_parse_complex(c: &mut Criterion) {
|
||||
let parser = QueryParser::default();
|
||||
|
||||
c.bench_function("parse_complex_mixed", |b| {
|
||||
b.iter(|| parser.parse(black_box("src name *.rs !test /lib/ status:modified")));
|
||||
});
|
||||
|
||||
c.bench_function("parse_glob", |b| {
|
||||
b.iter(|| parser.parse(black_box("**/*.rs")));
|
||||
});
|
||||
|
||||
c.bench_function("parse_multiple_constraints", |b| {
|
||||
b.iter(|| parser.parse(black_box("*.rs *.toml *.md !test !node_modules /src/")));
|
||||
});
|
||||
}
|
||||
|
||||
fn bench_parse_realistic_queries(c: &mut Criterion) {
|
||||
let parser = QueryParser::default();
|
||||
|
||||
let queries = vec![
|
||||
"file",
|
||||
"test",
|
||||
"mod.rs",
|
||||
"src/*.rs",
|
||||
"lib test",
|
||||
"*.rs !test",
|
||||
"src/lib/*.rs",
|
||||
"/src/ name",
|
||||
"status:modified *.rs",
|
||||
"type:rust test !node_modules",
|
||||
];
|
||||
|
||||
let mut group = c.benchmark_group("realistic_queries");
|
||||
for query in queries.iter() {
|
||||
group.throughput(Throughput::Bytes(query.len() as u64));
|
||||
group.bench_with_input(BenchmarkId::from_parameter(query), query, |b, q| {
|
||||
b.iter(|| parser.parse(black_box(q)));
|
||||
});
|
||||
}
|
||||
group.finish();
|
||||
}
|
||||
|
||||
fn bench_parse_various_lengths(c: &mut Criterion) {
|
||||
let parser = QueryParser::default();
|
||||
|
||||
let short = "*.rs";
|
||||
let medium = "src name *.rs !test";
|
||||
let long = "src lib test name *.rs *.toml !node_modules !test /src/ /lib/ status:modified";
|
||||
let very_long =
|
||||
"a b c d e f g h i j k l m n o p q r s t u v w x y z *.rs *.toml *.md *.txt *.js";
|
||||
|
||||
let mut group = c.benchmark_group("query_lengths");
|
||||
|
||||
group.throughput(Throughput::Bytes(short.len() as u64));
|
||||
group.bench_with_input(BenchmarkId::new("short", short.len()), &short, |b, q| {
|
||||
b.iter(|| parser.parse(black_box(q)));
|
||||
});
|
||||
|
||||
group.throughput(Throughput::Bytes(medium.len() as u64));
|
||||
group.bench_with_input(BenchmarkId::new("medium", medium.len()), &medium, |b, q| {
|
||||
b.iter(|| parser.parse(black_box(q)));
|
||||
});
|
||||
|
||||
group.throughput(Throughput::Bytes(long.len() as u64));
|
||||
group.bench_with_input(BenchmarkId::new("long", long.len()), &long, |b, q| {
|
||||
b.iter(|| parser.parse(black_box(q)));
|
||||
});
|
||||
|
||||
group.throughput(Throughput::Bytes(very_long.len() as u64));
|
||||
group.bench_with_input(
|
||||
BenchmarkId::new("very_long", very_long.len()),
|
||||
&very_long,
|
||||
|b, q| {
|
||||
b.iter(|| parser.parse(black_box(q)));
|
||||
},
|
||||
);
|
||||
|
||||
group.finish();
|
||||
}
|
||||
|
||||
fn bench_config_comparison(c: &mut Criterion) {
|
||||
let file_picker = QueryParser::new(FileSearchConfig);
|
||||
let grep = QueryParser::new(GrepConfig);
|
||||
|
||||
let query = "src name *.rs !test";
|
||||
|
||||
let mut group = c.benchmark_group("config_comparison");
|
||||
|
||||
group.bench_function("file_picker_config", |b| {
|
||||
b.iter(|| file_picker.parse(black_box(query)));
|
||||
});
|
||||
|
||||
group.bench_function("grep_config", |b| {
|
||||
b.iter(|| grep.parse(black_box(query)));
|
||||
});
|
||||
|
||||
group.finish();
|
||||
}
|
||||
|
||||
fn bench_constraint_types(c: &mut Criterion) {
|
||||
let parser = QueryParser::default();
|
||||
|
||||
let mut group = c.benchmark_group("constraint_types");
|
||||
|
||||
group.bench_function("extension", |b| {
|
||||
b.iter(|| parser.parse(black_box("*.rs")));
|
||||
});
|
||||
|
||||
group.bench_function("glob", |b| {
|
||||
b.iter(|| parser.parse(black_box("**/*.rs")));
|
||||
});
|
||||
|
||||
group.bench_function("exclude", |b| {
|
||||
b.iter(|| parser.parse(black_box("!test")));
|
||||
});
|
||||
|
||||
group.bench_function("path_segment", |b| {
|
||||
b.iter(|| parser.parse(black_box("/src/")));
|
||||
});
|
||||
|
||||
group.bench_function("git_status", |b| {
|
||||
b.iter(|| parser.parse(black_box("status:modified")));
|
||||
});
|
||||
|
||||
group.bench_function("file_type", |b| {
|
||||
b.iter(|| parser.parse(black_box("type:rust")));
|
||||
});
|
||||
|
||||
group.finish();
|
||||
}
|
||||
|
||||
fn bench_worst_case(c: &mut Criterion) {
|
||||
let parser = QueryParser::default();
|
||||
|
||||
// Worst case: many constraints that all need to be checked
|
||||
let worst_case = "a b c d e f g h i j k l m n o p q r s t u v w x y z";
|
||||
|
||||
c.bench_function("worst_case_many_text_tokens", |b| {
|
||||
b.iter(|| parser.parse(black_box(worst_case)));
|
||||
});
|
||||
|
||||
// Many constraints
|
||||
let many_constraints = "*.rs *.toml *.md *.txt *.js *.ts *.jsx *.tsx *.vue *.svelte";
|
||||
|
||||
c.bench_function("worst_case_many_constraints", |b| {
|
||||
b.iter(|| parser.parse(black_box(many_constraints)));
|
||||
});
|
||||
}
|
||||
|
||||
criterion_group!(
|
||||
benches,
|
||||
bench_parse_simple,
|
||||
bench_parse_complex,
|
||||
bench_parse_realistic_queries,
|
||||
bench_parse_various_lengths,
|
||||
bench_config_comparison,
|
||||
bench_constraint_types,
|
||||
bench_worst_case,
|
||||
);
|
||||
|
||||
criterion_main!(benches);
|
||||
+42
-12
@@ -1,16 +1,46 @@
|
||||
fn main() {
|
||||
// Opt-in cfg for the long-running randomized stress tests
|
||||
// used by tests/fuzz_git_watcher_stress.rs
|
||||
println!("cargo::rustc-check-cfg=cfg(stress)");
|
||||
|
||||
// When the `zlob` feature is enabled (Zig-compiled C library):
|
||||
// On Windows MSVC, explicitly link the C runtime libraries.
|
||||
// This is needed because Zig-compiled static libraries (zlob) don't emit
|
||||
// /DEFAULTLIB directives for the MSVC CRT. Without this, symbols like
|
||||
// strcmp, memcpy, memchr etc. from vendored C libraries (libgit2, lmdb)
|
||||
// are unresolved when linking the cdylib.
|
||||
//
|
||||
// We link both msvcrt (classic CRT) and ucrt (Universal CRT where memchr,
|
||||
// strcmp etc. live on newer MSVC/ARM64 targets).
|
||||
let target = std::env::var("TARGET").unwrap_or_default();
|
||||
if target.contains("windows") && target.contains("msvc") {
|
||||
println!("cargo:rustc-link-lib=msvcrt");
|
||||
println!("cargo:rustc-link-lib=ucrt");
|
||||
println!("cargo:rustc-link-lib=vcruntime");
|
||||
// Zig-compiled static libraries don't emit /DEFAULTLIB directives for the
|
||||
// MSVC CRT, so symbols like strcmp, memcpy etc. would be unresolved.
|
||||
if std::env::var("CARGO_FEATURE_ZLOB").is_ok() {
|
||||
if !zig_available() {
|
||||
panic!(
|
||||
"The `zlob` feature is enabled but Zig is not installed. \
|
||||
Install Zig (https://ziglang.org/download/) or build without \
|
||||
`--features zlob`."
|
||||
);
|
||||
}
|
||||
|
||||
let target = std::env::var("TARGET").unwrap_or_default();
|
||||
if target.contains("windows") && target.contains("msvc") {
|
||||
println!("cargo:rustc-link-lib=msvcrt");
|
||||
println!("cargo:rustc-link-lib=ucrt");
|
||||
println!("cargo:rustc-link-lib=vcruntime");
|
||||
}
|
||||
} else if std::env::var("CARGO_PRIMARY_PACKAGE").is_ok() && zig_available() {
|
||||
// Hint: if Zig is available but the zlob feature wasn't enabled,
|
||||
// let the developer know they can get faster glob matching.
|
||||
// Only emit this hint when this crate is the primary package to
|
||||
// avoid noisy warnings for downstream consumers.
|
||||
println!(
|
||||
"cargo:warning=Zig detected but `zlob` feature is not enabled. \
|
||||
Build with `--features zlob` for faster glob matching."
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/// Probe the system for a working Zig installation.
|
||||
fn zig_available() -> bool {
|
||||
std::process::Command::new("zig")
|
||||
.arg("version")
|
||||
.stdout(std::process::Stdio::null())
|
||||
.stderr(std::process::Stdio::null())
|
||||
.status()
|
||||
.map(|s| s.success())
|
||||
.unwrap_or(false)
|
||||
}
|
||||
|
||||
@@ -1,137 +1,413 @@
|
||||
use crate::FILE_PICKER;
|
||||
use crate::error::Error;
|
||||
use crate::file_picker::FilePicker;
|
||||
use crate::file_picker::{FFFMode, MAX_OVERFLOW_FILES};
|
||||
use crate::git::GitStatusCache;
|
||||
use crate::shared::{SharedFilePicker, SharedFrecency};
|
||||
use crate::sort_buffer::sort_with_buffer;
|
||||
use git2::Repository;
|
||||
use notify::event::{AccessKind, AccessMode};
|
||||
use notify::{Config, EventKind, RecursiveMode};
|
||||
use notify_debouncer_full::{
|
||||
DebounceEventResult, DebouncedEvent, RecommendedCache, new_debouncer_opt,
|
||||
};
|
||||
use notify::{Config, EventKind, EventKindMask, RecursiveMode};
|
||||
use notify_debouncer_full::{DebounceEventResult, DebouncedEvent, NoCache, new_debouncer_opt};
|
||||
use parking_lot::Mutex;
|
||||
use std::path::{Path, PathBuf};
|
||||
use std::sync::{Arc, Mutex};
|
||||
use std::sync::Arc;
|
||||
use std::sync::mpsc;
|
||||
use std::time::Duration;
|
||||
use tracing::{Level, error, info, warn};
|
||||
use tracing::{Level, debug, error, info, warn};
|
||||
|
||||
type Debouncer = notify_debouncer_full::Debouncer<notify::RecommendedWatcher, RecommendedCache>;
|
||||
type Debouncer = notify_debouncer_full::Debouncer<notify::RecommendedWatcher, NoCache>;
|
||||
|
||||
/// Owns the file-system watcher and guarantees that all background threads
|
||||
/// are fully joined before `stop()` / `Drop` returns.
|
||||
pub struct BackgroundWatcher {
|
||||
debouncer: Arc<Mutex<Option<Debouncer>>>,
|
||||
watch_tx: Option<mpsc::Sender<PathBuf>>,
|
||||
owner_thread: Option<std::thread::JoinHandle<()>>,
|
||||
}
|
||||
|
||||
const DEBOUNCE_TIMEOUT: Duration = Duration::from_millis(250);
|
||||
const DEBOUNCE_TIMEOUT: Duration = Duration::from_millis(50);
|
||||
const MAX_PATHS_THRESHOLD: usize = 1024;
|
||||
const MAX_SELECTIVE_WATCH_DIRS: usize = 100;
|
||||
/// On macOS, each `watch()` call creates a separate FSEventStream. When the
|
||||
/// number of directories exceeds this threshold we fall back to a single
|
||||
/// recursive watch to avoid exhausting the per-process stream limit.
|
||||
const MAX_MACOS_NONRECURSIVE_WATCHES: usize = 4096;
|
||||
/// Minimum seconds between frecency tracks of the same file in AI mode.
|
||||
/// Prevents score inflation from rapid burst edits by AI agents.
|
||||
const AI_MODE_COOLDOWN_SECS: u64 = 5 * 60;
|
||||
|
||||
impl BackgroundWatcher {
|
||||
pub fn new(base_path: PathBuf, git_workdir: Option<PathBuf>) -> Result<Self, Error> {
|
||||
pub fn new(
|
||||
base_path: PathBuf,
|
||||
git_workdir: Option<PathBuf>,
|
||||
shared_picker: SharedFilePicker,
|
||||
shared_frecency: SharedFrecency,
|
||||
mode: FFFMode,
|
||||
) -> Result<Self, Error> {
|
||||
info!(
|
||||
"Initializing background watcher for path: {}",
|
||||
base_path.display()
|
||||
"Initializing background watcher for path: {}, mode: {:?}",
|
||||
base_path.display(),
|
||||
mode,
|
||||
);
|
||||
|
||||
let debouncer = Self::create_debouncer(base_path, git_workdir)?;
|
||||
// Refuse to watch the filesystem root or the user's home directory.
|
||||
// These are prone to high-volume event churn (editor temp files,
|
||||
// browser caches, log rotations) which inflates the overflow arena
|
||||
// and, on macOS, can exhaust the per-process FSEvents stream limit.
|
||||
if base_path.parent().is_none()
|
||||
|| Some(base_path.as_os_str()) == dirs::home_dir().as_ref().map(|p| p.as_os_str())
|
||||
{
|
||||
return Err(Error::FilesystemRoot(base_path));
|
||||
}
|
||||
|
||||
// macOS: always use a single recursive FSEvent stream.
|
||||
//
|
||||
// Per-dir NonRecursive watches create one FSEvent stream per dir.
|
||||
// The per-process FSEvent cap is lower than expected in practice
|
||||
// (4096 per process, but FFF usually is running within code editors),
|
||||
// and each failed `watch()` after the cap blocks ~40 ms on kernel retry.
|
||||
// Yes we pay for filtering events on handler phase but it is usable
|
||||
//
|
||||
// macOS and Windows use a single recursive watch. FSEvents and
|
||||
// ReadDirectoryChangesW both support true kernel-level recursion
|
||||
// on one handle — per-dir NonRecursive watches burn streams/handles
|
||||
// for no benefit and, on Windows, have been observed to silently
|
||||
// drop Modify events for nested paths.
|
||||
//
|
||||
// Linux keeps the per-dir NonRecursive strategy: inotify has no
|
||||
// kernel-level recursion, so Recursive here would still register
|
||||
// one watch per subdir but without the ignored-dir filtering we
|
||||
// get by iterating `picker.for_each_dir` ourselves.
|
||||
let use_recursive = cfg!(any(target_os = "macos", target_os = "windows"));
|
||||
|
||||
let (watch_tx, watch_rx) = mpsc::channel::<PathBuf>();
|
||||
let watch_tx_for_debouncer = watch_tx.clone();
|
||||
|
||||
let owner_weak_picker = shared_picker.weaken();
|
||||
let owner_frecency = shared_frecency.clone();
|
||||
let owner_git_workdir = git_workdir.clone();
|
||||
|
||||
let debouncer = Self::create_debouncer(
|
||||
base_path,
|
||||
git_workdir,
|
||||
shared_picker,
|
||||
shared_frecency,
|
||||
mode,
|
||||
use_recursive,
|
||||
watch_tx_for_debouncer,
|
||||
)?;
|
||||
|
||||
info!("Background file watcher initialized successfully");
|
||||
|
||||
// debouncer is shared with the owner thread, once it's dropped the thread is closed
|
||||
let debouncer = Arc::new(Mutex::new(Some(debouncer)));
|
||||
// Only the Linux per-dir-watch branch needs this clone; on other
|
||||
// platforms the owner thread never touches the debouncer.
|
||||
#[cfg(target_os = "linux")]
|
||||
let owner_debouncer = Arc::clone(&debouncer);
|
||||
|
||||
let owner_thread = std::thread::Builder::new()
|
||||
.name("fff-watcher-own".into())
|
||||
.spawn(move || {
|
||||
while let Ok(dir) = watch_rx.recv() {
|
||||
// if the picker is dropped we do need to exit the loop
|
||||
let Some(strong_picker) = owner_weak_picker.upgrade() else {
|
||||
break;
|
||||
};
|
||||
|
||||
// Only inotify (Linux) has no kernel-level recursion, so
|
||||
// it's the only platform that needs a per-subdir watch to
|
||||
// be registered at runtime. macOS FSEvents and Windows
|
||||
// ReadDirectoryChangesW are already watching recursively
|
||||
// from the base path (see `create_debouncer`), and
|
||||
// registering a second overlapping stream there produces
|
||||
// duplicate/out-of-order events.
|
||||
#[cfg(target_os = "linux")]
|
||||
{
|
||||
// Register the new directory with the debouncer, then
|
||||
// drop the mutex BEFORE doing picker-side work — see
|
||||
// the comment on `BackgroundWatcher::stop` for the
|
||||
// lock-ordering rationale.
|
||||
let mut guard = owner_debouncer.lock();
|
||||
let Some(debouncer) = guard.as_mut() else {
|
||||
break;
|
||||
};
|
||||
|
||||
if let Err(e) = debouncer.watch(&dir, RecursiveMode::NonRecursive) {
|
||||
warn!(
|
||||
?e,
|
||||
dir = %dir.display(),
|
||||
"Failed to init watcher for new directory"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
track_files_from_new_directories(
|
||||
&dir,
|
||||
&strong_picker,
|
||||
&owner_frecency,
|
||||
&owner_git_workdir,
|
||||
);
|
||||
|
||||
// Transient strong ref drops here, back
|
||||
// to weak-only before the next `recv()`.
|
||||
}
|
||||
|
||||
tracing::info!("Background watcher is stopped");
|
||||
})
|
||||
.expect("failed to spawn fff-watcher-owner thread");
|
||||
|
||||
Ok(Self {
|
||||
debouncer: Arc::new(Mutex::new(Some(debouncer))),
|
||||
debouncer,
|
||||
watch_tx: Some(watch_tx),
|
||||
owner_thread: Some(owner_thread),
|
||||
})
|
||||
}
|
||||
|
||||
fn create_debouncer(
|
||||
base_path: PathBuf,
|
||||
git_workdir: Option<PathBuf>,
|
||||
shared_picker: SharedFilePicker,
|
||||
shared_frecency: SharedFrecency,
|
||||
mode: FFFMode,
|
||||
use_recursive: bool,
|
||||
watch_tx: mpsc::Sender<PathBuf>,
|
||||
) -> Result<Debouncer, Error> {
|
||||
// do not follow symlinks as then notifiers spawns a bunch of events for symlinked
|
||||
// files that could be git ignored, we have to property differentiate those and if
|
||||
// the file was edited through a
|
||||
let config = Config::default().with_follow_symlinks(false);
|
||||
let config = Config::default()
|
||||
// do not follow symlinks as then notifiers spawns a bunch of events for symlinked
|
||||
// files that could be git ignored, we have to property differentiate those and if
|
||||
// the file was edited through a
|
||||
.with_follow_symlinks(false)
|
||||
// only the actual modification events, ignore the open syscals that we can generate by
|
||||
// our own grep calls and preview window rendering
|
||||
.with_event_kinds(EventKindMask::CORE);
|
||||
|
||||
// `use_recursive` was decided by the caller from a cheap size hint,
|
||||
// so the event-handler closure can capture it directly.
|
||||
//
|
||||
// The closure lives on the debouncer's internal event thread
|
||||
// for as long as the debouncer exists — i.e. the full
|
||||
// lifetime of `BackgroundWatcher`. Capturing a strong
|
||||
// `SharedFilePicker` here would re-introduce the Arc cycle
|
||||
// we just broke with `owner_picker`'s `downgrade()` above.
|
||||
// Capture a weak handle instead and upgrade per-batch.
|
||||
let git_workdir_for_handler = git_workdir.clone();
|
||||
let shared_picker_for_watching = shared_picker.clone();
|
||||
let event_picker = shared_picker.weaken();
|
||||
let mut debouncer = new_debouncer_opt(
|
||||
DEBOUNCE_TIMEOUT,
|
||||
Some(DEBOUNCE_TIMEOUT / 2), // tick rate for the event span
|
||||
{
|
||||
move |result: DebounceEventResult| match result {
|
||||
Ok(events) => {
|
||||
handle_debounced_events(events, &git_workdir);
|
||||
// Upgrade just long enough to drive one
|
||||
// debounced batch. Failure means every
|
||||
// external `SharedFilePicker` has already
|
||||
// dropped and teardown is already underway.
|
||||
let Some(strong_picker) = event_picker.upgrade() else {
|
||||
return;
|
||||
};
|
||||
|
||||
let new_dirs = handle_debounced_events(
|
||||
events,
|
||||
&git_workdir_for_handler,
|
||||
&strong_picker,
|
||||
&shared_frecency,
|
||||
mode,
|
||||
);
|
||||
|
||||
// every new directory creates had to be reflected in the picker state
|
||||
for dir in new_dirs {
|
||||
if let Err(e) = watch_tx.send(dir) {
|
||||
warn!(?e, "Failed to send directory update error");
|
||||
}
|
||||
}
|
||||
}
|
||||
Err(errors) => {
|
||||
error!("File watcher errors: {:?}", errors);
|
||||
}
|
||||
}
|
||||
},
|
||||
RecommendedCache::new(),
|
||||
// There is an issue with recommended cache implementation on macos
|
||||
// it keeps track of all the files added to the watcher which is not a problem
|
||||
// for us because any rename to the file will anyway require the removing from the
|
||||
// ordedred index and adding it back with the new name
|
||||
NoCache::new(),
|
||||
config,
|
||||
)?;
|
||||
|
||||
// Watch only non-ignored directories to avoid flooding the OS event buffer.
|
||||
// On macOS, FSEvents has a fixed-size kernel buffer — watching huge gitignored
|
||||
// directories like `target/` in rust causes buffer overflow, which drops real source file
|
||||
// events. Instead we watch the root non-recursively (for top-level file changes
|
||||
// and new directory detection) and each non-ignored subdirectory recursively.
|
||||
let watch_dirs = collect_non_ignored_dirs(&base_path);
|
||||
// Watching strategy:
|
||||
//
|
||||
// For small-to-medium repos we watch each indexed directory individually
|
||||
// (NonRecursive). This avoids receiving events for gitignored paths like
|
||||
// node_modules/ and keeps the event volume low.
|
||||
//
|
||||
// On macOS, each `watch()` call creates a separate FSEventStream. Large
|
||||
// repos (e.g. Chromium with 487K+ files) can have tens of thousands of
|
||||
// directories, which exhausts the per-process FSEvents stream limit and
|
||||
// causes "unable to start FSEvent stream" errors. When the directory
|
||||
// count exceeds the threshold we fall back to a single Recursive watch
|
||||
// on the base path. FSEvents handles this efficiently with one kernel
|
||||
// stream for the entire subtree. Gitignored paths are already filtered
|
||||
// in the event handler via `should_include_file()`.
|
||||
//
|
||||
// On Linux (inotify), RecursiveMode::Recursive creates one kernel watch
|
||||
// per subdirectory *including* gitignored ones, wasting file descriptors.
|
||||
// The per-directory NonRecursive approach is always used on Linux.
|
||||
//
|
||||
// New directories created at runtime are detected via Create events on
|
||||
// the parent and dynamically added by the owner thread via watch_tx.
|
||||
|
||||
if watch_dirs.len() > MAX_SELECTIVE_WATCH_DIRS {
|
||||
tracing::warn!(
|
||||
"Too many non-ignored directories ({}/{}) can't efficiently watch them",
|
||||
watch_dirs.len(),
|
||||
MAX_SELECTIVE_WATCH_DIRS
|
||||
);
|
||||
if use_recursive {
|
||||
debouncer.watch(base_path.as_path(), RecursiveMode::Recursive)?;
|
||||
info!(
|
||||
"File watcher initialized with single recursive watch on {} \
|
||||
(exceeded threshold of {})",
|
||||
base_path.display(),
|
||||
MAX_MACOS_NONRECURSIVE_WATCHES,
|
||||
);
|
||||
} else {
|
||||
debouncer.watch(base_path.as_path(), RecursiveMode::NonRecursive)?;
|
||||
|
||||
for dir in &watch_dirs {
|
||||
match debouncer.watch(dir.as_path(), RecursiveMode::Recursive) {
|
||||
Ok(()) => {}
|
||||
Err(e) => {
|
||||
// Non-fatal: directory may have been removed between discovery and watch
|
||||
warn!("Failed to watch directory {}: {}", dir.display(), e);
|
||||
// Stream watch-dir registration directly under the picker
|
||||
// read lock. Only Linux (inotify) reaches this branch —
|
||||
// macOS always takes the recursive path above. `inotify`'s
|
||||
// `inotify_add_watch()` is fast-fail: on ENOSPC it returns
|
||||
// immediately, no kernel retry loop, so holding the read
|
||||
// lock across the stream is O(ms) even for large repos.
|
||||
//
|
||||
// Abort the loop after a run of failures. Once ENOSPC hits,
|
||||
// further calls won't succeed until the user raises
|
||||
// `fs.inotify.max_user_watches`, so there's no value in
|
||||
// continuing.
|
||||
const MAX_CONSECUTIVE_WATCH_FAILURES: usize = 16;
|
||||
|
||||
let mut watched = 0usize;
|
||||
let mut consecutive_failures = 0usize;
|
||||
let mut aborted_early = false;
|
||||
|
||||
if let Some(guard) = shared_picker_for_watching.read().ok()
|
||||
&& let Some(picker) = guard.as_ref()
|
||||
{
|
||||
use std::ops::ControlFlow;
|
||||
picker.for_each_dir(|dir| {
|
||||
match debouncer.watch(dir, RecursiveMode::NonRecursive) {
|
||||
Ok(()) => {
|
||||
watched += 1;
|
||||
consecutive_failures = 0;
|
||||
ControlFlow::Continue(())
|
||||
}
|
||||
Err(e) => {
|
||||
consecutive_failures += 1;
|
||||
if consecutive_failures <= 4 {
|
||||
warn!("Failed to watch directory {}: {}", dir.display(), e);
|
||||
}
|
||||
|
||||
if consecutive_failures >= MAX_CONSECUTIVE_WATCH_FAILURES {
|
||||
warn!(
|
||||
consecutive_failures,
|
||||
watched,
|
||||
"Aborting NonRecursive watch loop — per-process \
|
||||
watch cap exhausted, further dirs would just burn \
|
||||
kernel time for no coverage"
|
||||
);
|
||||
aborted_early = true;
|
||||
ControlFlow::Break(())
|
||||
} else {
|
||||
ControlFlow::Continue(())
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
info!(
|
||||
"File watcher initialized for {} directories (NonRecursive) under {} (aborted_early={})",
|
||||
watched,
|
||||
base_path.display(),
|
||||
aborted_early,
|
||||
);
|
||||
}
|
||||
|
||||
info!(
|
||||
"File watcher initialized for {} directories under {}",
|
||||
watch_dirs.len(),
|
||||
base_path.display()
|
||||
);
|
||||
// The .git directory is excluded from the file list but we still need
|
||||
// to observe changes that affect git status (staging, unstaging,
|
||||
// committing, branch switches, merges, etc.).
|
||||
// When using recursive mode the base watch already covers .git/,
|
||||
// but these targeted watches are cheap (at most 3 extra streams)
|
||||
// and ensure we catch status changes even if the recursive backend
|
||||
// coalesces or delays .git events.
|
||||
watch_git_status_paths(&mut debouncer, git_workdir.as_ref());
|
||||
|
||||
Ok(debouncer)
|
||||
}
|
||||
|
||||
pub fn stop(&self) {
|
||||
if let Ok(Some(debouncer)) = self.debouncer.lock().map(|mut debouncer| debouncer.take()) {
|
||||
drop(debouncer);
|
||||
info!("Background file watcher stopped successfully");
|
||||
} else {
|
||||
error!("Failed to stop background watcher");
|
||||
/// Signal the watcher to shut down without blocking on its worker
|
||||
/// threads. Safe to call from any context, including while holding
|
||||
/// the [`SharedFilePicker`] write lock.
|
||||
///
|
||||
/// Both the debouncer's internal event loop and our owner thread
|
||||
/// may call `SharedFilePicker::write()` inside their handlers. A
|
||||
/// blocking join here would deadlock against a caller that already
|
||||
/// holds that lock (e.g. `stop_background_monitor` under a
|
||||
/// `shared_picker.write()` guard). Instead we:
|
||||
///
|
||||
/// * drop the `watch_tx` Sender — the owner thread's
|
||||
/// `watch_rx.recv()` returns `Err` and the thread exits at
|
||||
/// its next `recv`.
|
||||
/// * call `debouncer.stop_nonblocking()` — signals the debouncer
|
||||
/// event loop to exit on its next tick and drops the watcher,
|
||||
/// closing the FSEvent / inotify / ReadDirectoryChangesW stream.
|
||||
/// * detach both `JoinHandle`s.
|
||||
///
|
||||
/// In-flight handler invocations finish on their own (at most one
|
||||
/// more batch) once the caller releases any locks they hold.
|
||||
pub fn stop(&mut self) {
|
||||
self.watch_tx.take();
|
||||
if let Some(debouncer) = self.debouncer.lock().take() {
|
||||
debouncer.stop_nonblocking();
|
||||
}
|
||||
|
||||
self.owner_thread.take();
|
||||
|
||||
info!("Background file watcher stop signaled");
|
||||
}
|
||||
|
||||
/// Queue a non-recursive watch registration on `dir`.
|
||||
///
|
||||
/// The owner thread is always blocked on `watch_rx.recv()`, so
|
||||
/// the `send()` here wakes it immediately via the channel's
|
||||
/// condvar — no external unpark needed.
|
||||
///
|
||||
/// Returns `false` once `stop()` has dropped our `Sender` — any
|
||||
/// further request is silently discarded.
|
||||
pub(crate) fn request_watch_dir(&self, dir: PathBuf) -> bool {
|
||||
match self.watch_tx.as_ref() {
|
||||
Some(tx) => tx.send(dir).is_ok(),
|
||||
None => false,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl Drop for BackgroundWatcher {
|
||||
fn drop(&mut self) {
|
||||
if let Ok(mut debouncer_guard) = self.debouncer.lock() {
|
||||
if let Some(debouncer) = debouncer_guard.take() {
|
||||
drop(debouncer);
|
||||
}
|
||||
} else {
|
||||
error!("Failed to acquire debouncer lock to drop");
|
||||
}
|
||||
self.stop();
|
||||
}
|
||||
}
|
||||
|
||||
#[tracing::instrument(name = "fs_events", skip(events), level = Level::DEBUG)]
|
||||
fn handle_debounced_events(events: Vec<DebouncedEvent>, git_workdir: &Option<PathBuf>) {
|
||||
#[tracing::instrument(name = "fs_events", skip(events, shared_picker, shared_frecency), level = Level::DEBUG)]
|
||||
fn handle_debounced_events(
|
||||
events: Vec<DebouncedEvent>,
|
||||
git_workdir: &Option<PathBuf>,
|
||||
shared_picker: &SharedFilePicker,
|
||||
shared_frecency: &SharedFrecency,
|
||||
mode: FFFMode,
|
||||
) -> Vec<PathBuf> {
|
||||
// this will be called very often, we have to minimiy the lock time for file picker
|
||||
let repo = git_workdir.as_ref().and_then(|p| Repository::open(p).ok());
|
||||
let mut need_full_rescan = false;
|
||||
let mut need_full_git_rescan = false;
|
||||
let mut paths_to_remove = Vec::new();
|
||||
let mut dirs_to_remove: Vec<PathBuf> = Vec::new();
|
||||
let mut paths_to_add_or_modify = Vec::new();
|
||||
let mut new_dirs_to_watch = Vec::new();
|
||||
let mut affected_paths_count = 0usize;
|
||||
|
||||
for debounced_event in &events {
|
||||
@@ -175,14 +451,48 @@ fn handle_debounced_events(events: Vec<DebouncedEvent>, git_workdir: &Option<Pat
|
||||
need_full_git_rescan = true;
|
||||
}
|
||||
|
||||
if !should_include_file(path, &repo) {
|
||||
if is_git_file(path) {
|
||||
continue;
|
||||
}
|
||||
|
||||
if !path.exists() {
|
||||
// Use a combination of event kind and filesystem state to decide
|
||||
// whether a path is an addition/modification or a removal.
|
||||
//
|
||||
// We cannot rely on `path.exists()` alone because:
|
||||
// - A freshly created file might not be visible yet (race).
|
||||
// - macOS FSEvents uses Modify(Name(Any)) for both rename-in
|
||||
// and rename-out, so we must stat the path to disambiguate.
|
||||
//
|
||||
// We cannot rely on event kind alone because:
|
||||
// - Remove events are not always emitted (macOS often sends
|
||||
// Modify(Name(Any)) instead of Remove).
|
||||
let is_removal = matches!(debounced_event.event.kind, EventKind::Remove(_));
|
||||
|
||||
// Directory-level remove: both fsevents and inotify delivers a single
|
||||
// `Remove(Folder)` event for a whole directory tree (e.g.
|
||||
// after `git reset --hard` wipes a dir full of staged-but-
|
||||
// uncommitted files).
|
||||
let is_folder_removal = matches!(
|
||||
debounced_event.event.kind,
|
||||
EventKind::Remove(notify::event::RemoveKind::Folder)
|
||||
);
|
||||
|
||||
if is_folder_removal {
|
||||
dirs_to_remove.push(path.to_path_buf());
|
||||
} else if is_removal || !path.exists() {
|
||||
paths_to_remove.push(path.as_path());
|
||||
} else if path.is_dir() {
|
||||
// New directory — collect it so the caller can register a
|
||||
// watcher. No filesystem scanning: files that arrive later
|
||||
// will be handled by the newly registered watch.
|
||||
if !is_path_ignored(path, &repo) {
|
||||
new_dirs_to_watch.push(path.to_path_buf());
|
||||
}
|
||||
} else {
|
||||
paths_to_add_or_modify.push(path.as_path());
|
||||
// For additions/modifications, still filter gitignored files.
|
||||
if should_include_file(path, &repo) {
|
||||
paths_to_add_or_modify.push(path.as_path());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -204,8 +514,10 @@ fn handle_debounced_events(events: Vec<DebouncedEvent>, git_workdir: &Option<Pat
|
||||
|
||||
if need_full_rescan {
|
||||
info!(?affected_paths_count, "Triggering full rescan");
|
||||
trigger_full_rescan();
|
||||
return;
|
||||
if let Err(e) = shared_picker.trigger_full_rescan_async(shared_frecency) {
|
||||
error!("Failed to trigger full rescan: {:?}", e);
|
||||
}
|
||||
return Vec::new();
|
||||
}
|
||||
|
||||
// It's important to get the allocated sort
|
||||
@@ -215,150 +527,304 @@ fn handle_debounced_events(events: Vec<DebouncedEvent>, git_workdir: &Option<Pat
|
||||
paths_to_add_or_modify.dedup_by(|a, b| a.as_os_str().eq(b.as_os_str()));
|
||||
|
||||
info!(
|
||||
"Event processing summary: {} to remove, {} to add/modify",
|
||||
"Event processing summary: {} to remove, {} dirs to remove, {} to add/modify, {} new dirs",
|
||||
paths_to_remove.len(),
|
||||
paths_to_add_or_modify.len()
|
||||
dirs_to_remove.len(),
|
||||
paths_to_add_or_modify.len(),
|
||||
new_dirs_to_watch.len()
|
||||
);
|
||||
|
||||
let Some(repo) = repo.as_ref() else {
|
||||
info!("No git repo, skipping git status updates");
|
||||
return;
|
||||
};
|
||||
|
||||
if need_full_git_rescan {
|
||||
info!("Triggering full git rescan");
|
||||
|
||||
if let Err(e) = FilePicker::refresh_git_status_global() {
|
||||
error!("Failed to refresh git status: {:?}", e);
|
||||
}
|
||||
|
||||
return;
|
||||
if paths_to_remove.is_empty()
|
||||
&& dirs_to_remove.is_empty()
|
||||
&& paths_to_add_or_modify.is_empty()
|
||||
&& !need_full_git_rescan
|
||||
{
|
||||
debug!("No file index changes to apply");
|
||||
return new_dirs_to_watch;
|
||||
}
|
||||
|
||||
if paths_to_remove.is_empty() && paths_to_add_or_modify.is_empty() {
|
||||
return;
|
||||
}
|
||||
let mut files_to_update_git_status = Vec::new();
|
||||
let mut need_full_rescan = false;
|
||||
let mut overflow_count = 0;
|
||||
|
||||
let files_to_update_git_status = {
|
||||
let Ok(mut file_picker_guard) = FILE_PICKER.write() else {
|
||||
if !paths_to_remove.is_empty()
|
||||
|| !dirs_to_remove.is_empty()
|
||||
|| !paths_to_add_or_modify.is_empty()
|
||||
{
|
||||
debug!(
|
||||
"Applying file index changes: {} to remove, {} dirs to remove, {} to add/modify",
|
||||
paths_to_remove.len(),
|
||||
dirs_to_remove.len(),
|
||||
paths_to_add_or_modify.len(),
|
||||
);
|
||||
|
||||
let Ok(mut guard) = shared_picker.write() else {
|
||||
error!("Failed to acquire file picker write lock");
|
||||
return;
|
||||
return new_dirs_to_watch;
|
||||
};
|
||||
|
||||
let Some(ref mut picker) = *file_picker_guard else {
|
||||
let Some(ref mut picker) = *guard else {
|
||||
error!("File picker not initialized");
|
||||
return;
|
||||
return new_dirs_to_watch;
|
||||
};
|
||||
|
||||
// Apply file removals
|
||||
for path in paths_to_remove {
|
||||
picker.remove_file_by_path(path);
|
||||
// No need to invalidate mmap — the FileItem (and its mmap) is dropped
|
||||
for dir in &dirs_to_remove {
|
||||
let count = picker.remove_all_files_in_dir(dir);
|
||||
debug!("remove_all_files_in_dir({:?}) -> {} files", dir, count);
|
||||
}
|
||||
|
||||
// Apply file additions/modifications and collect paths for git status update
|
||||
let mut files_to_update_git_status = Vec::with_capacity(paths_to_add_or_modify.len());
|
||||
for path in paths_to_add_or_modify {
|
||||
// on_create_or_modify clears the mmap internally when modified time changes
|
||||
if let Some(file) = picker.on_create_or_modify(path) {
|
||||
files_to_update_git_status.push(file.path.clone());
|
||||
for path in &paths_to_remove {
|
||||
let removed = picker.remove_file_by_path(path);
|
||||
debug!("remove_file_by_path({:?}) -> {}", path, removed);
|
||||
}
|
||||
|
||||
files_to_update_git_status.reserve(paths_to_add_or_modify.len());
|
||||
for path in &paths_to_add_or_modify {
|
||||
if picker.handle_create_or_modify(path).is_some() {
|
||||
files_to_update_git_status.push(path.to_path_buf());
|
||||
} else {
|
||||
need_full_rescan = true;
|
||||
}
|
||||
}
|
||||
|
||||
files_to_update_git_status
|
||||
};
|
||||
overflow_count = picker.get_overflow_files().len();
|
||||
}
|
||||
|
||||
info!(
|
||||
"Fetching git status for {} files",
|
||||
files_to_update_git_status.len()
|
||||
files_updated = files_to_update_git_status.len(),
|
||||
overflow_count, "File index changes applied",
|
||||
);
|
||||
|
||||
let status = match GitStatusCache::git_status_for_paths(repo, &files_to_update_git_status) {
|
||||
Ok(status) => status,
|
||||
Err(e) => {
|
||||
tracing::error!(?e, "Failed to query git statue");
|
||||
return;
|
||||
if need_full_rescan || overflow_count > MAX_OVERFLOW_FILES {
|
||||
info!("Watcher faced limit of index overflow. Triggering rescan");
|
||||
if let Err(e) = shared_picker.trigger_full_rescan_async(shared_frecency) {
|
||||
error!("Failed to trigger full rescan: {:?}", e);
|
||||
}
|
||||
};
|
||||
|
||||
// only lock the picker for theshortest possitble time
|
||||
if let Ok(mut file_picker_guard) = FILE_PICKER.write()
|
||||
&& let Some(ref mut picker) = *file_picker_guard
|
||||
{
|
||||
if let Err(e) = picker.update_git_statuses(status) {
|
||||
error!("Failed to update git statuses: {:?}", e);
|
||||
} else {
|
||||
info!("Successfully updated git statuses in picker");
|
||||
}
|
||||
} else {
|
||||
error!("Failed to acquire picker lock for git status update");
|
||||
}
|
||||
|
||||
// AI mode: auto-track frecency for all modified/created files.
|
||||
// Uses a 5-minute cooldown per file to prevent score inflation from rapid
|
||||
// burst edits (AI agents often edit the same file many times in minutes).
|
||||
// This runs after apply_changes so the picker write lock is released.
|
||||
if mode.is_ai() && !paths_to_add_or_modify.is_empty() {
|
||||
let mut tracked_count = 0usize;
|
||||
if let Ok(frecency_guard) = shared_frecency.read()
|
||||
&& let Some(ref frecency) = *frecency_guard
|
||||
{
|
||||
for path in &paths_to_add_or_modify {
|
||||
// Skip if this file was tracked less than 5 minutes ago
|
||||
let should_track = match frecency.seconds_since_last_access(path) {
|
||||
Ok(Some(secs)) => secs >= AI_MODE_COOLDOWN_SECS,
|
||||
Ok(None) => true, // Never tracked before
|
||||
Err(_) => true, // DB error, track anyway
|
||||
};
|
||||
if !should_track {
|
||||
continue;
|
||||
}
|
||||
|
||||
if let Err(e) = frecency.track_access(path) {
|
||||
error!("Failed to track frecency for {:?}: {:?}", path, e);
|
||||
} else {
|
||||
tracked_count += 1;
|
||||
}
|
||||
}
|
||||
if tracked_count > 0 {
|
||||
info!("AI mode: tracked frecency for {} files", tracked_count);
|
||||
}
|
||||
}
|
||||
|
||||
// Update in-memory frecency scores for tracked files
|
||||
if tracked_count > 0
|
||||
&& let Ok(mut picker_guard) = shared_picker.write()
|
||||
&& let Some(ref mut picker) = *picker_guard
|
||||
&& let Ok(frecency_guard) = shared_frecency.read()
|
||||
&& let Some(ref frecency) = *frecency_guard
|
||||
{
|
||||
for path in &paths_to_add_or_modify {
|
||||
let _ = picker.update_single_file_frecency(path, frecency);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// do not try to update the paths if we anyway going to rescan everything from scratch
|
||||
if !need_full_rescan && (need_full_git_rescan || !files_to_update_git_status.is_empty()) {
|
||||
let git_workdir = repo
|
||||
.as_ref()
|
||||
.map(|r| r.workdir().unwrap_or_else(|| r.path()).to_path_buf());
|
||||
|
||||
let shared_picker = shared_picker.clone();
|
||||
let shared_frecency = shared_frecency.clone();
|
||||
|
||||
// git status query even with a pathspec could be really slow, if we do this syncrhronously
|
||||
// within the event handler, we actually risk of forming a snow ball of conflicting events
|
||||
crate::file_picker::BACKGROUND_THREAD_POOL.spawn(move || {
|
||||
let Some(git_path) = git_workdir else { return };
|
||||
let Ok(repo) = Repository::open(&git_path) else {
|
||||
error!("Failed to open git repo for async status update");
|
||||
return;
|
||||
};
|
||||
|
||||
if need_full_git_rescan && !need_full_rescan {
|
||||
info!("Async: triggering full git rescan");
|
||||
if let Err(e) = shared_picker.refresh_git_status(&shared_frecency) {
|
||||
error!("Failed to refresh git status: {:?}", e);
|
||||
}
|
||||
}
|
||||
|
||||
if !files_to_update_git_status.is_empty() {
|
||||
let status = match GitStatusCache::git_status_for_paths(
|
||||
&repo,
|
||||
&files_to_update_git_status,
|
||||
) {
|
||||
Ok(s) => s,
|
||||
Err(e) => {
|
||||
error!("Failed to query git status: {:?}", e);
|
||||
return;
|
||||
}
|
||||
};
|
||||
|
||||
if let Ok(mut guard) = shared_picker.write()
|
||||
&& let Some(ref mut picker) = *guard
|
||||
{
|
||||
if let Err(e) = picker.update_git_statuses(status, &shared_frecency) {
|
||||
error!("Failed to update git statuses: {:?}", e);
|
||||
} else {
|
||||
info!("Async: git statuses updated");
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
new_dirs_to_watch
|
||||
}
|
||||
|
||||
fn trigger_full_rescan() {
|
||||
info!("Triggering full filesystem rescan");
|
||||
|
||||
// Note: no need to clear mmaps — they are backed by the kernel page cache
|
||||
// and automatically reflect file changes. Old FileItems (and their mmaps)
|
||||
// are dropped when the picker rebuilds its file list.
|
||||
|
||||
let Ok(mut file_picker_guard) = FILE_PICKER.write() else {
|
||||
error!("Failed to acquire file picker write lock for full rescan");
|
||||
/// After registering a watch on a newly created directory, list its
|
||||
/// immediate children and add any files to the picker.
|
||||
fn track_files_from_new_directories(
|
||||
dir: &Path,
|
||||
shared_picker: &SharedFilePicker,
|
||||
shared_frecency: &SharedFrecency,
|
||||
git_workdir: &Option<PathBuf>,
|
||||
) {
|
||||
let Ok(entries) = std::fs::read_dir(dir) else {
|
||||
return;
|
||||
};
|
||||
|
||||
let Some(ref mut picker) = *file_picker_guard else {
|
||||
error!("File picker not initialized, cannot trigger rescan");
|
||||
return;
|
||||
};
|
||||
let repo = git_workdir.as_ref().and_then(|p| Repository::open(p).ok());
|
||||
let mut files_to_add = Vec::new();
|
||||
|
||||
if let Err(e) = picker.trigger_rescan() {
|
||||
error!("Failed to trigger full rescan: {:?}", e);
|
||||
} else {
|
||||
info!("Full filesystem rescan completed successfully");
|
||||
for entry in entries.flatten() {
|
||||
if entry.file_type().is_ok_and(|ft| ft.is_file()) {
|
||||
let path = entry.path();
|
||||
if should_include_file(&path, &repo) {
|
||||
files_to_add.push(path);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if files_to_add.is_empty() {
|
||||
return;
|
||||
}
|
||||
|
||||
{
|
||||
let Ok(mut guard) = shared_picker.write() else {
|
||||
return;
|
||||
};
|
||||
|
||||
let Some(ref mut picker) = *guard else {
|
||||
return;
|
||||
};
|
||||
|
||||
for path in &files_to_add {
|
||||
picker.handle_create_or_modify(path);
|
||||
}
|
||||
}
|
||||
|
||||
if let Some(repo) = repo.as_ref() {
|
||||
let status = match GitStatusCache::git_status_for_paths(repo, &files_to_add) {
|
||||
Ok(status) => status,
|
||||
Err(e) => {
|
||||
tracing::error!(?e, "inject_existing_files: git status query failed");
|
||||
return;
|
||||
}
|
||||
};
|
||||
|
||||
if let Ok(mut guard) = shared_picker.write()
|
||||
&& let Some(ref mut picker) = *guard
|
||||
&& let Err(e) = picker.update_git_statuses(status, shared_frecency)
|
||||
{
|
||||
error!("inject_existing_files: failed to update git statuses: {e:?}");
|
||||
}
|
||||
}
|
||||
|
||||
debug!(
|
||||
"Injected {} existing files from new directory {}",
|
||||
files_to_add.len(),
|
||||
dir.display(),
|
||||
);
|
||||
}
|
||||
|
||||
fn should_include_file(path: &Path, repo: &Option<Repository>) -> bool {
|
||||
if !path.is_file() || is_git_file(path) {
|
||||
// Directories are not indexed — only regular files (and symlinks to files).
|
||||
if path.is_dir() {
|
||||
return false;
|
||||
}
|
||||
|
||||
repo.as_ref()
|
||||
.is_some_and(|repo| repo.is_path_ignored(path) == Ok(false))
|
||||
match repo.as_ref() {
|
||||
Some(repo) => repo.is_path_ignored(path) != Ok(true),
|
||||
None => {
|
||||
// No git repo — apply basic sanity filters.
|
||||
// Hidden directories are skipped by the watcher setup (hidden(true)),
|
||||
// but events can still arrive for files in known non-code directories.
|
||||
!is_non_code_directory(path)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn is_non_code_directory(path: &Path) -> bool {
|
||||
crate::ignore::is_non_code_directory(path)
|
||||
}
|
||||
|
||||
#[inline]
|
||||
fn is_git_file(path: &Path) -> bool {
|
||||
fn is_path_ignored(path: &Path, repo: &Option<Repository>) -> bool {
|
||||
match repo.as_ref() {
|
||||
Some(repo) => repo.is_path_ignored(path) == Ok(true),
|
||||
None => is_non_code_directory(path),
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn is_git_file(path: &Path) -> bool {
|
||||
path.components()
|
||||
.any(|component| component.as_os_str() == ".git")
|
||||
}
|
||||
|
||||
pub fn is_dotgit_change_affecting_status(changed: &Path, repo: &Option<Repository>) -> bool {
|
||||
fn is_dotgit_change_affecting_status(changed: &Path, repo: &Option<Repository>) -> bool {
|
||||
let Some(repo) = repo.as_ref() else {
|
||||
return false;
|
||||
};
|
||||
|
||||
let git_dir = repo.path();
|
||||
|
||||
if let Ok(rel) = changed.strip_prefix(git_dir) {
|
||||
if rel.starts_with("objects") || rel.starts_with("logs") || rel.starts_with("hooks") {
|
||||
return false;
|
||||
}
|
||||
if rel == Path::new("index") || rel == Path::new("index.lock") {
|
||||
if let Ok(path_in_git_dir) = changed.strip_prefix(git_dir) {
|
||||
// Only react to changes that rewrite the worktree state: commits,
|
||||
// staging, checkouts, merges, conflict resolution. Ref-only updates
|
||||
// under refs/ (fetch, push, tag writes, pack-refs) do not change
|
||||
// which files are modified/untracked, so we deliberately skip them —
|
||||
// watching refs/ recursively would cost one inotify watch per ref
|
||||
// namespace on repos with many branches/remotes.
|
||||
if path_in_git_dir == Path::new("index") || path_in_git_dir == Path::new("index.lock") {
|
||||
return true;
|
||||
}
|
||||
if rel == Path::new("HEAD") {
|
||||
if path_in_git_dir == Path::new("HEAD") {
|
||||
return true;
|
||||
}
|
||||
if rel.starts_with("refs") || rel == Path::new("packed-refs") {
|
||||
return true;
|
||||
}
|
||||
if rel == Path::new("info/exclude") || rel == Path::new("info/sparse-checkout") {
|
||||
if path_in_git_dir == Path::new("info/exclude")
|
||||
|| path_in_git_dir == Path::new("info/sparse-checkout")
|
||||
{
|
||||
return true;
|
||||
}
|
||||
|
||||
if let Some(fname) = rel.file_name().and_then(|f| f.to_str())
|
||||
if let Some(fname) = path_in_git_dir.file_name().and_then(|f| f.to_str())
|
||||
&& matches!(fname, "MERGE_HEAD" | "CHERRY_PICK_HEAD" | "REVERT_HEAD")
|
||||
{
|
||||
return true;
|
||||
@@ -375,37 +841,31 @@ fn is_ignore_definition_path(path: &Path) -> bool {
|
||||
)
|
||||
}
|
||||
|
||||
/// Collects immediate non-ignored subdirectories of `base_path` using the `ignore` crate
|
||||
/// to respect .gitignore, .ignore, and global gitignore rules. This is used to set up
|
||||
/// selective file watching — only non-ignored directories get a recursive watcher,
|
||||
/// preventing gitignored directories like `target/` from flooding the OS event buffer.
|
||||
fn collect_non_ignored_dirs(base_path: &Path) -> Vec<PathBuf> {
|
||||
use ignore::WalkBuilder;
|
||||
fn watch_git_status_paths(debouncer: &mut Debouncer, git_workdir: Option<&PathBuf>) {
|
||||
let Some(workdir) = git_workdir else {
|
||||
return;
|
||||
};
|
||||
|
||||
let walker = WalkBuilder::new(base_path)
|
||||
.hidden(false)
|
||||
.git_ignore(true)
|
||||
.git_exclude(true)
|
||||
.git_global(true)
|
||||
.ignore(true)
|
||||
.follow_links(false)
|
||||
.max_depth(Some(1))
|
||||
.build();
|
||||
|
||||
let mut dirs = Vec::new();
|
||||
for entry in walker {
|
||||
let Ok(entry) = entry else { continue };
|
||||
let path = entry.path();
|
||||
|
||||
// Skip the root directory itself
|
||||
if path == base_path {
|
||||
continue;
|
||||
}
|
||||
|
||||
if path.is_dir() && !is_git_file(path) {
|
||||
dirs.push(path.to_path_buf());
|
||||
}
|
||||
let git_dir = workdir.join(".git");
|
||||
if !git_dir.is_dir() {
|
||||
return;
|
||||
}
|
||||
|
||||
dirs
|
||||
// Watch .git/ non-recursively to catch top-level files:
|
||||
// index, index.lock, HEAD, MERGE_HEAD, CHERRY_PICK_HEAD, REVERT_HEAD.
|
||||
// We intentionally do NOT watch refs/ — individual ref updates don't
|
||||
// affect worktree status, and a recursive watch there blows up inotify
|
||||
// watch counts on repos with many branches/remotes/tags.
|
||||
if let Err(e) = debouncer.watch(&git_dir, RecursiveMode::NonRecursive) {
|
||||
warn!("Failed to watch .git directory: {}", e);
|
||||
return;
|
||||
}
|
||||
|
||||
// Watch info/ non-recursively for exclude and sparse-checkout
|
||||
let info_dir = git_dir.join("info");
|
||||
if info_dir.is_dir()
|
||||
&& let Err(e) = debouncer.watch(&info_dir, RecursiveMode::NonRecursive)
|
||||
{
|
||||
warn!("Failed to watch .git/info: {}", e);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,983 @@
|
||||
use ahash::AHashMap;
|
||||
use rayon::iter::{IndexedParallelIterator, ParallelIterator};
|
||||
use rayon::slice::ParallelSlice;
|
||||
use std::cell::UnsafeCell;
|
||||
use std::sync::OnceLock;
|
||||
use std::sync::atomic::{AtomicU16, AtomicUsize, Ordering};
|
||||
|
||||
/// Maximum number of distinct bigrams tracked in the inverted index.
|
||||
/// 95 printable ASCII chars (32..=126) after lowercasing → ~70 distinct → 4900 possible.
|
||||
/// We cap at 5000 to cover all printable bigrams with margin.
|
||||
/// 5000 columns × 62.5KB (500k files) = 305MB. For 50k files: 30MB.
|
||||
const MAX_BIGRAM_COLUMNS: usize = 5000;
|
||||
|
||||
/// Sentinel value: bigram has no allocated column.
|
||||
const NO_COLUMN: u16 = u16::MAX;
|
||||
|
||||
/// Temporary sync dense builder for the bigram index.
|
||||
/// Builds from the many threads reading file contents in parallel
|
||||
pub struct BigramIndexBuilder {
|
||||
// we use lookup as atomics only in the builder because it is filled by the rayon threads
|
||||
// the actual index uses pure u16 for the allocations
|
||||
lookup: Vec<AtomicU16>,
|
||||
/// Flat bitset data, materialised on first use.
|
||||
col_data: OnceLock<UnsafeCell<Box<[u64]>>>,
|
||||
next_column: AtomicU16,
|
||||
words: usize,
|
||||
file_count: usize,
|
||||
populated: AtomicUsize,
|
||||
}
|
||||
|
||||
// SAFETY: `col_data`'s interior mutability is coordinated via disjoint
|
||||
// `word_idx` ranges (word-aligned file partitioning in the driver), so
|
||||
// concurrent access is safe despite the `UnsafeCell`. See builder doc.
|
||||
unsafe impl Sync for BigramIndexBuilder {}
|
||||
|
||||
impl BigramIndexBuilder {
|
||||
pub fn new(file_count: usize) -> Self {
|
||||
let words = file_count.div_ceil(64);
|
||||
let mut lookup = Vec::with_capacity(65536);
|
||||
lookup.resize_with(65536, || AtomicU16::new(NO_COLUMN));
|
||||
Self {
|
||||
lookup,
|
||||
col_data: OnceLock::new(),
|
||||
next_column: AtomicU16::new(0),
|
||||
words,
|
||||
file_count,
|
||||
populated: AtomicUsize::new(0),
|
||||
}
|
||||
}
|
||||
|
||||
/// Lazily materialise the full `MAX_BIGRAM_COLUMNS * words` bitset
|
||||
/// on first access.
|
||||
#[inline(always)]
|
||||
fn col_data_cell(&self) -> &UnsafeCell<Box<[u64]>> {
|
||||
self.col_data.get_or_init(|| {
|
||||
let total = MAX_BIGRAM_COLUMNS * self.words;
|
||||
UnsafeCell::new(vec![0u64; total].into_boxed_slice())
|
||||
})
|
||||
}
|
||||
|
||||
/// Raw pointer to the start of the bitset slab. Used for in-place
|
||||
/// `|=` writes under the partitioning invariant.
|
||||
#[inline(always)]
|
||||
fn col_data_ptr(&self) -> *mut u64 {
|
||||
unsafe { (*self.col_data_cell().get()).as_mut_ptr() }
|
||||
}
|
||||
|
||||
#[inline]
|
||||
fn get_or_alloc_column(&self, key: u16) -> u16 {
|
||||
let current = self.lookup[key as usize].load(Ordering::Relaxed);
|
||||
if current != NO_COLUMN {
|
||||
return current;
|
||||
}
|
||||
let new_col = self.next_column.fetch_add(1, Ordering::Relaxed);
|
||||
if new_col >= MAX_BIGRAM_COLUMNS as u16 {
|
||||
return NO_COLUMN;
|
||||
}
|
||||
|
||||
match self.lookup[key as usize].compare_exchange(
|
||||
NO_COLUMN,
|
||||
new_col,
|
||||
Ordering::Relaxed,
|
||||
Ordering::Relaxed,
|
||||
) {
|
||||
Ok(_) => new_col,
|
||||
Err(existing) => existing,
|
||||
}
|
||||
}
|
||||
|
||||
/// SAFETY: caller must not access the same `word_idx` slot from
|
||||
/// another thread concurrently. Partitioning in
|
||||
/// `file_picker::build_bigram_index` enforces this.
|
||||
#[inline(always)]
|
||||
unsafe fn column_word_ptr(&self, col: u16, word_idx: usize) -> *mut u64 {
|
||||
unsafe {
|
||||
self.col_data_ptr()
|
||||
.add(col as usize * self.words + word_idx)
|
||||
}
|
||||
}
|
||||
|
||||
/// Test/bench accessor for a column's raw bitset words. Assumes the
|
||||
/// caller has joined all writers (no concurrent mutation).
|
||||
#[cfg(test)]
|
||||
fn column_bitset(&self, col: u16) -> &[u64] {
|
||||
let start = col as usize * self.words;
|
||||
let slab = unsafe { &*self.col_data_cell().get() };
|
||||
&slab[start..start + self.words]
|
||||
}
|
||||
|
||||
// `pub` (via `#[doc(hidden)]`) only so the criterion bench can drive
|
||||
// `add_file_content` directly. External consumers should use
|
||||
// `build_bigram_index` instead.
|
||||
///
|
||||
/// SAFETY: concurrent callers must partition `file_idx` by
|
||||
/// word-aligned ranges so that `file_idx / 64` never collides across
|
||||
/// threads. The `file_picker::build_bigram_index` driver enforces
|
||||
/// this via `par_chunks` with a word-aligned chunk size.
|
||||
#[doc(hidden)]
|
||||
pub fn add_file_content(&self, skip_builder: &Self, file_idx: usize, content: &[u8]) {
|
||||
if content.len() < 2 {
|
||||
return;
|
||||
}
|
||||
|
||||
debug_assert!(file_idx < self.file_count);
|
||||
let word_idx = file_idx / 64;
|
||||
let bit_mask = 1u64 << (file_idx % 64);
|
||||
|
||||
// Stack-local dedup bitsets: 1024 × u64 = 8 KB each, covers all 65536
|
||||
// bigram keys with margin. Has to fit in L1 cache.
|
||||
let mut seen_consec = [0u64; 1024];
|
||||
let mut seen_skip = [0u64; 1024];
|
||||
|
||||
// Normalise each byte as we stream and carry a 2-byte history
|
||||
// across iterations so each input byte is normalised exactly once
|
||||
// even though it participates in up to three bigrams (as `cur`,
|
||||
// then `prev`, then `skip_prev`). Benchmarked against a NEON
|
||||
// pre-pass variant — the pre-pass needs a heap scratch per call,
|
||||
// which kills throughput unless content is gigantic. Inline
|
||||
// normalisation is the faster choice for realistic file sizes.
|
||||
let bytes = content;
|
||||
let len = bytes.len();
|
||||
|
||||
let mut n0 = normalize_byte_scalar(bytes[0]);
|
||||
let mut n1 = normalize_byte_scalar(bytes[1]);
|
||||
|
||||
if n0 != u16::MAX && n1 != u16::MAX {
|
||||
let key = (n0 << 8) | n1;
|
||||
self.record_bigram(&mut seen_consec, key, word_idx, bit_mask);
|
||||
}
|
||||
|
||||
for &b in &bytes[2..len] {
|
||||
let cur = normalize_byte_scalar(b);
|
||||
if cur != u16::MAX {
|
||||
if n1 != u16::MAX {
|
||||
let key = (n1 << 8) | cur;
|
||||
self.record_bigram(&mut seen_consec, key, word_idx, bit_mask);
|
||||
}
|
||||
if n0 != u16::MAX {
|
||||
let key = (n0 << 8) | cur;
|
||||
skip_builder.record_bigram(&mut seen_skip, key, word_idx, bit_mask);
|
||||
}
|
||||
}
|
||||
n0 = n1;
|
||||
n1 = cur;
|
||||
}
|
||||
|
||||
self.populated.fetch_add(1, Ordering::Relaxed);
|
||||
skip_builder.populated.fetch_add(1, Ordering::Relaxed);
|
||||
}
|
||||
|
||||
/// Mark `key` as present for the file whose column-word is `word_idx`
|
||||
/// and bit position is `bit_mask`, de-duplicating via the caller-owned
|
||||
/// `seen` bitmap so we only touch the shared column slab at most once
|
||||
/// per unique bigram per file.
|
||||
///
|
||||
/// SAFETY: under the partitioning invariant on `add_file_content`
|
||||
/// the `word_idx` slot this touches is owned exclusively by the
|
||||
/// current thread, so a plain `|=` through the raw pointer is
|
||||
/// race-free (no atomic RMW needed).
|
||||
#[inline(always)]
|
||||
fn record_bigram(&self, seen: &mut [u64; 1024], key: u16, word_idx: usize, bit_mask: u64) {
|
||||
let k = key as usize;
|
||||
let w = k >> 6;
|
||||
let bit = 1u64 << (k & 63);
|
||||
if seen[w] & bit == 0 {
|
||||
seen[w] |= bit;
|
||||
let col = self.get_or_alloc_column(key);
|
||||
if col != NO_COLUMN {
|
||||
unsafe {
|
||||
let p = self.column_word_ptr(col, word_idx);
|
||||
*p |= bit_mask;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub fn is_ready(&self) -> bool {
|
||||
self.populated.load(Ordering::Relaxed) > 0
|
||||
}
|
||||
|
||||
pub fn columns_used(&self) -> u16 {
|
||||
self.next_column
|
||||
.load(Ordering::Relaxed)
|
||||
.min(MAX_BIGRAM_COLUMNS as u16)
|
||||
}
|
||||
|
||||
/// Compress the dense builder into a compact `BigramFilter`.
|
||||
///
|
||||
/// Retains columns where the bigram appears in ≥`min_density_pct`% (or
|
||||
/// the default ~3.1% heuristic when `None`) and <90% of indexed files.
|
||||
/// Sparse columns carry too little data to justify their memory;
|
||||
/// ubiquitous columns (≥90%) are nearly all-ones and barely filter.
|
||||
#[inline(always)]
|
||||
pub fn compress(self, min_density_pct: Option<u32>) -> BigramFilter {
|
||||
let cols = self.columns_used() as usize;
|
||||
let words = self.words;
|
||||
let file_count = self.file_count;
|
||||
let populated = self.populated.load(Ordering::Relaxed);
|
||||
let dense_bytes = words * 8; // cost of one dense column
|
||||
|
||||
let old_lookup = self.lookup;
|
||||
// If no file ever populated content, col_data was never
|
||||
// materialised. Treat as empty — every column falls through.
|
||||
let col_data: Option<Box<[u64]>> = self.col_data.into_inner().map(UnsafeCell::into_inner);
|
||||
|
||||
let mut lookup: Vec<u16> = vec![NO_COLUMN; 65536];
|
||||
let mut dense_data: Vec<u64> = Vec::with_capacity(cols * words);
|
||||
let mut dense_count: usize = 0;
|
||||
|
||||
if let Some(col_data) = col_data.as_deref() {
|
||||
for key in 0..65536usize {
|
||||
let old_col = old_lookup[key].load(Ordering::Relaxed);
|
||||
if old_col == NO_COLUMN || old_col as usize >= cols {
|
||||
continue;
|
||||
}
|
||||
|
||||
let col_start = old_col as usize * words;
|
||||
let bitset = &col_data[col_start..col_start + words];
|
||||
|
||||
// count set bits to decide if this column is worth keeping.
|
||||
let mut popcount = 0u32;
|
||||
for &word in bitset.iter().take(words) {
|
||||
popcount += word.count_ones();
|
||||
}
|
||||
|
||||
// drop bigrams appearing in too few files
|
||||
let not_to_rare = if let Some(min_pct) = min_density_pct {
|
||||
// Percentage-based: require ≥ min_pct% of populated files.
|
||||
populated > 0 && (popcount as usize) * 100 >= populated * min_pct as usize
|
||||
} else {
|
||||
// Default: popcount ≥ words × 2 (~3.1% of files).
|
||||
(popcount as usize * 4) >= dense_bytes
|
||||
};
|
||||
|
||||
if !not_to_rare {
|
||||
continue;
|
||||
}
|
||||
|
||||
// Drop ubiquitous bigrams — columns ≥90% ones carry almost no
|
||||
// filtering power and just waste memory + AND cycles.
|
||||
if populated > 0 && (popcount as usize) * 10 >= populated * 9 {
|
||||
continue;
|
||||
}
|
||||
|
||||
let dense_idx = dense_count as u16;
|
||||
lookup[key] = dense_idx;
|
||||
dense_count += 1;
|
||||
|
||||
dense_data.extend_from_slice(bitset);
|
||||
}
|
||||
}
|
||||
|
||||
BigramFilter {
|
||||
lookup,
|
||||
dense_data,
|
||||
dense_count,
|
||||
words,
|
||||
file_count,
|
||||
populated,
|
||||
skip_index: None,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
unsafe impl Send for BigramIndexBuilder {}
|
||||
|
||||
/// Inverted bigram index with optional "skip-1" extension
|
||||
/// Copmressed into bitset for minimal usage, the layout of this struct actually matters
|
||||
#[derive(Debug)]
|
||||
pub struct BigramFilter {
|
||||
lookup: Vec<u16>,
|
||||
/// Flat buffer of all dense column data laid out at fixed stride `words`.
|
||||
/// Column `i` starts at `i * words`.
|
||||
dense_data: Vec<u64>, // do not try to change this to u8 it has to be wordsize
|
||||
dense_count: usize,
|
||||
words: usize,
|
||||
file_count: usize,
|
||||
populated: usize,
|
||||
/// Optional skip-1 bigram index (stride 2). Built from character pairs
|
||||
/// at distance 2, e.g. "ABCDE" → (A,C),(B,D),(C,E). ANDead with the
|
||||
/// consecutive bigram candidates during query to dramatically reduce
|
||||
/// false positives.
|
||||
skip_index: Option<Box<BigramFilter>>,
|
||||
}
|
||||
|
||||
/// SIMD-friendly bitwise AND of two equal-length bitsets.
|
||||
// Auto vectorized (don't touch)
|
||||
#[inline]
|
||||
fn bitset_and(result: &mut [u64], bitset: &[u64]) {
|
||||
result
|
||||
.iter_mut()
|
||||
.zip(bitset.iter())
|
||||
.for_each(|(r, b)| *r &= *b);
|
||||
}
|
||||
|
||||
impl BigramFilter {
|
||||
/// AND the posting lists for all query bigrams (consecutive + skip).
|
||||
/// Returns None if no query bigrams are tracked.
|
||||
pub fn query(&self, pattern: &[u8]) -> Option<Vec<u64>> {
|
||||
if pattern.len() < 2 {
|
||||
return None;
|
||||
}
|
||||
|
||||
let mut result = vec![u64::MAX; self.words];
|
||||
if !self.file_count.is_multiple_of(64) {
|
||||
let last = self.words - 1;
|
||||
result[last] = (1u64 << (self.file_count % 64)) - 1;
|
||||
}
|
||||
|
||||
let words = self.words;
|
||||
let mut has_filter = false;
|
||||
|
||||
let mut prev = pattern[0];
|
||||
for &b in &pattern[1..] {
|
||||
if (32..=126).contains(&prev) && (32..=126).contains(&b) {
|
||||
let key = (prev.to_ascii_lowercase() as u16) << 8 | b.to_ascii_lowercase() as u16;
|
||||
let col = self.lookup[key as usize];
|
||||
if col != NO_COLUMN {
|
||||
let offset = col as usize * words;
|
||||
// SAFETY: compress() guarantees offset + words <= dense_data.len()
|
||||
let slice = unsafe { self.dense_data.get_unchecked(offset..offset + words) };
|
||||
bitset_and(&mut result, slice);
|
||||
has_filter = true;
|
||||
}
|
||||
}
|
||||
prev = b;
|
||||
}
|
||||
|
||||
// strid-1 bigrams
|
||||
if let Some(skip) = &self.skip_index
|
||||
&& pattern.len() >= 3
|
||||
&& let Some(skip_candidates) = skip.query_skip(pattern)
|
||||
{
|
||||
bitset_and(&mut result, &skip_candidates);
|
||||
has_filter = true;
|
||||
}
|
||||
|
||||
has_filter.then_some(result)
|
||||
}
|
||||
|
||||
/// Query using stride-2 bigrams from the pattern.
|
||||
/// For "ABCDE" queries with keys (A,C), (B,D), (C,E).
|
||||
fn query_skip(&self, pattern: &[u8]) -> Option<Vec<u64>> {
|
||||
let mut result = vec![u64::MAX; self.words];
|
||||
if !self.file_count.is_multiple_of(64) {
|
||||
let last = self.words - 1;
|
||||
result[last] = (1u64 << (self.file_count % 64)) - 1;
|
||||
}
|
||||
|
||||
let words = self.words;
|
||||
let mut has_filter = false;
|
||||
|
||||
for i in 0..pattern.len().saturating_sub(2) {
|
||||
let a = pattern[i];
|
||||
let b = pattern[i + 2];
|
||||
if (32..=126).contains(&a) && (32..=126).contains(&b) {
|
||||
let key = (a.to_ascii_lowercase() as u16) << 8 | b.to_ascii_lowercase() as u16;
|
||||
let col = self.lookup[key as usize];
|
||||
if col != NO_COLUMN {
|
||||
let offset = col as usize * words;
|
||||
let slice = unsafe { self.dense_data.get_unchecked(offset..offset + words) };
|
||||
bitset_and(&mut result, slice);
|
||||
has_filter = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
has_filter.then_some(result)
|
||||
}
|
||||
|
||||
/// Attach a skip-1 bigram index for tighter candidate filtering.
|
||||
pub fn set_skip_index(&mut self, skip: BigramFilter) {
|
||||
self.skip_index = Some(Box::new(skip));
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub fn is_candidate(candidates: &[u64], file_idx: usize) -> bool {
|
||||
let word = file_idx / 64;
|
||||
let bit = file_idx % 64;
|
||||
word < candidates.len() && candidates[word] & (1u64 << bit) != 0
|
||||
}
|
||||
|
||||
pub fn count_candidates(candidates: &[u64]) -> usize {
|
||||
candidates.iter().map(|w| w.count_ones() as usize).sum()
|
||||
}
|
||||
|
||||
pub fn is_ready(&self) -> bool {
|
||||
self.populated > 0
|
||||
}
|
||||
|
||||
pub fn file_count(&self) -> usize {
|
||||
self.file_count
|
||||
}
|
||||
|
||||
pub fn columns_used(&self) -> usize {
|
||||
self.dense_count
|
||||
}
|
||||
|
||||
/// Total heap bytes used by this index (lookup + dense data + skip).
|
||||
pub fn heap_bytes(&self) -> usize {
|
||||
let lookup_bytes = self.lookup.len() * std::mem::size_of::<u16>();
|
||||
let dense_bytes = self.dense_data.len() * std::mem::size_of::<u64>();
|
||||
let skip_bytes = self.skip_index.as_ref().map_or(0, |s| s.heap_bytes());
|
||||
lookup_bytes + dense_bytes + skip_bytes
|
||||
}
|
||||
|
||||
/// Check whether a bigram key is present in this index.
|
||||
pub fn has_key(&self, key: u16) -> bool {
|
||||
self.lookup[key as usize] != NO_COLUMN
|
||||
}
|
||||
|
||||
/// Raw lookup table (65536 entries mapping bigram key → column index).
|
||||
pub fn lookup(&self) -> &[u16] {
|
||||
&self.lookup
|
||||
}
|
||||
|
||||
/// Flat dense bitset data at fixed stride `words`.
|
||||
pub fn dense_data(&self) -> &[u64] {
|
||||
&self.dense_data
|
||||
}
|
||||
|
||||
/// Number of u64 words per column (= ceil(file_count / 64)).
|
||||
pub fn words(&self) -> usize {
|
||||
self.words
|
||||
}
|
||||
|
||||
/// Number of dense columns retained after compression.
|
||||
pub fn dense_count(&self) -> usize {
|
||||
self.dense_count
|
||||
}
|
||||
|
||||
/// Number of files that contributed content to the index.
|
||||
pub fn populated(&self) -> usize {
|
||||
self.populated
|
||||
}
|
||||
|
||||
/// Reference to the optional skip-1 bigram sub-index.
|
||||
pub fn skip_index(&self) -> Option<&BigramFilter> {
|
||||
self.skip_index.as_deref()
|
||||
}
|
||||
|
||||
/// Create a new bigram filter from the internal data
|
||||
pub fn new(
|
||||
lookup: Vec<u16>,
|
||||
dense_data: Vec<u64>,
|
||||
dense_count: usize,
|
||||
words: usize,
|
||||
file_count: usize,
|
||||
populated: usize,
|
||||
) -> Self {
|
||||
Self {
|
||||
lookup,
|
||||
dense_data,
|
||||
dense_count,
|
||||
words,
|
||||
file_count,
|
||||
populated,
|
||||
skip_index: None,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Map a single input byte to its normalised form used by the bigram
|
||||
/// builder: `u16::MAX` when not printable ASCII (outside `32..=126`),
|
||||
/// otherwise the lowercased byte value in `0..=126`. The `u16::MAX`
|
||||
/// sentinel can never collide with a printable-ASCII byte so the consumer
|
||||
/// can test `!= u16::MAX` without false positives.
|
||||
///
|
||||
/// Branchless and `#[inline(always)]`: LLVM lifts the ASCII-range check
|
||||
/// and the conditional-lowercase OR into a handful of instructions per
|
||||
/// call, so calling this inside a hot loop matches a hand-unrolled
|
||||
/// equivalent.
|
||||
#[inline(always)]
|
||||
fn normalize_byte_scalar(b: u8) -> u16 {
|
||||
let printable = b.wrapping_sub(32) <= 94;
|
||||
// Branchless lowercase: OR 0x20 iff byte is in 'A'..='Z'.
|
||||
let lower = b | ((b.wrapping_sub(b'A') < 26) as u8 * 0x20);
|
||||
if printable { lower as u16 } else { u16::MAX }
|
||||
}
|
||||
|
||||
pub fn extract_bigrams(content: &[u8]) -> Vec<u16> {
|
||||
if content.len() < 2 {
|
||||
return Vec::new();
|
||||
}
|
||||
// Use a flat bitset (65536 bits = 8 KB) for dedup — faster than HashSet.
|
||||
let mut seen = vec![0u64; 1024]; // 1024 * 64 = 65536 bits
|
||||
let mut bigrams = Vec::new();
|
||||
|
||||
let mut prev = content[0];
|
||||
for &b in &content[1..] {
|
||||
if (32..=126).contains(&prev) && (32..=126).contains(&b) {
|
||||
let key = (prev.to_ascii_lowercase() as u16) << 8 | b.to_ascii_lowercase() as u16;
|
||||
let word = key as usize / 64;
|
||||
let bit = 1u64 << (key as usize % 64);
|
||||
if seen[word] & bit == 0 {
|
||||
seen[word] |= bit;
|
||||
bigrams.push(key);
|
||||
}
|
||||
}
|
||||
prev = b;
|
||||
}
|
||||
bigrams
|
||||
}
|
||||
|
||||
/// Modified and added files store their own bigram sets. Deleted files are
|
||||
/// tombstoned in a bitset so they can be excluded from base query results.
|
||||
/// This overlay is updated by the background watcher on every file event
|
||||
/// and cleared when the base index is rebuilt.
|
||||
#[derive(Debug)]
|
||||
pub struct BigramOverlay {
|
||||
/// Per-file bigram sets for files modified since the base was built.
|
||||
/// Key = file index in the base `Vec<FileItem>`.
|
||||
modified: AHashMap<usize, Vec<u16>>,
|
||||
|
||||
/// Tombstone bitset — one bit per base file. Set bits are excluded
|
||||
/// from base query results.
|
||||
tombstones: Vec<u64>,
|
||||
|
||||
/// Original files count this overlay was created for.
|
||||
base_file_count: usize,
|
||||
}
|
||||
|
||||
impl BigramOverlay {
|
||||
pub(crate) fn new(base_file_count: usize) -> Self {
|
||||
let words = base_file_count.div_ceil(64);
|
||||
Self {
|
||||
modified: AHashMap::new(),
|
||||
tombstones: vec![0u64; words],
|
||||
base_file_count,
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn modify_file(&mut self, file_idx: usize, content: &[u8]) {
|
||||
self.modified.insert(file_idx, extract_bigrams(content));
|
||||
}
|
||||
|
||||
pub(crate) fn delete_file(&mut self, file_idx: usize) {
|
||||
if file_idx < self.base_file_count {
|
||||
let word = file_idx / 64;
|
||||
self.tombstones[word] |= 1u64 << (file_idx % 64);
|
||||
}
|
||||
self.modified.remove(&file_idx);
|
||||
}
|
||||
|
||||
/// Return base file indices of modified files whose bigrams match ALL
|
||||
/// of the given `pattern_bigrams`.
|
||||
pub(crate) fn query_modified(&self, pattern_bigrams: &[u16]) -> Vec<usize> {
|
||||
if pattern_bigrams.is_empty() {
|
||||
return self.modified.keys().copied().collect();
|
||||
}
|
||||
self.modified
|
||||
.iter()
|
||||
.filter_map(|(&file_idx, bigrams)| {
|
||||
pattern_bigrams
|
||||
.iter()
|
||||
.all(|pb| bigrams.contains(pb))
|
||||
.then_some(file_idx)
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Number of base files this overlay was created for.
|
||||
pub(crate) fn base_file_count(&self) -> usize {
|
||||
self.base_file_count
|
||||
}
|
||||
|
||||
/// Get the tombstone bitset for clearing base candidates.
|
||||
pub(crate) fn tombstones(&self) -> &[u64] {
|
||||
&self.tombstones
|
||||
}
|
||||
|
||||
/// Get all modified file indices (for conservative overlay merging when
|
||||
/// we can't extract precise bigrams, e.g. regex patterns).
|
||||
pub(crate) fn modified_indices(&self) -> Vec<usize> {
|
||||
self.modified.keys().copied().collect()
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) const MAX_INDEXABLE_FILE_SIZE: usize = 2 * 1024 * 1024;
|
||||
const BIGRAM_CHUNK_FILES: usize = 4 * 64;
|
||||
|
||||
/// Sparse-column cutoff for the skip-1 sub-index. Rare skip columns add
|
||||
/// little filtering power but ~25-30% of index memory, so we drop
|
||||
/// anything appearing in < 12 % of populated files.
|
||||
const SKIP_INDEX_MIN_DENSITY_PCT: u32 = 12;
|
||||
|
||||
thread_local! {
|
||||
/// Reusable read buffer that is allocated per thread and used for reading files
|
||||
static READ_BUF: std::cell::RefCell<Box<[u8]>> =
|
||||
std::cell::RefCell::new(vec![0u8; MAX_INDEXABLE_FILE_SIZE].into_boxed_slice());
|
||||
}
|
||||
|
||||
/// Reads bigram chunk, we *SHOULD NOT* use mmap cache here because bigram is built off-lock
|
||||
/// if the watcher thread tries to invalidate mmap during the borrow from it - UAB or segfaut
|
||||
///
|
||||
/// mmap should only be used by the locked version of grep which absolutely minimizes any riscs
|
||||
#[inline]
|
||||
fn read_bigram_chunk<'a>(
|
||||
file: &crate::types::FileItem,
|
||||
base_fd: libc::c_int,
|
||||
base_path: &std::path::Path,
|
||||
arena: crate::simd_path::ArenaPtr,
|
||||
buf: &'a mut [u8],
|
||||
path_buf: &mut [u8; crate::simd_path::PATH_BUF_SIZE],
|
||||
) -> Option<&'a [u8]> {
|
||||
let want = (file.size as usize).min(MAX_INDEXABLE_FILE_SIZE);
|
||||
let filled = file.read_trimmed_into_buf(base_fd, base_path, arena, path_buf, &mut buf[..want]);
|
||||
if filled == 0 {
|
||||
return None;
|
||||
}
|
||||
|
||||
let data = &buf[..filled];
|
||||
if crate::file_picker::detect_binary_content(data) {
|
||||
file.set_binary(true);
|
||||
return None;
|
||||
}
|
||||
Some(data)
|
||||
}
|
||||
|
||||
#[tracing::instrument(skip_all, name = "Building Bigram Index", level = tracing::Level::DEBUG)]
|
||||
pub(crate) fn build_bigram_index(
|
||||
files: &[crate::types::FileItem],
|
||||
base_path: &std::path::Path,
|
||||
arena: crate::simd_path::ArenaPtr,
|
||||
) -> BigramFilter {
|
||||
let builder = BigramIndexBuilder::new(files.len());
|
||||
let skip_builder = BigramIndexBuilder::new(files.len());
|
||||
|
||||
#[cfg(unix)]
|
||||
let base_fd: libc::c_int = open_base_dir_fd(base_path);
|
||||
#[cfg(not(unix))]
|
||||
let base_fd: i32 = -1;
|
||||
|
||||
// Always reads each file into the thread-local READ_BUF — never aliases the
|
||||
// persistent mmap cache. See `read_bigram_chunk` for the rationale: this
|
||||
// pass runs detached on the background pool without holding the picker
|
||||
// read lock, so a watcher event mutating a `FileItem` would race any
|
||||
// borrow we took from a cached `Mmap`.
|
||||
crate::file_picker::BACKGROUND_THREAD_POOL.install(|| {
|
||||
files
|
||||
.par_chunks(BIGRAM_CHUNK_FILES)
|
||||
.enumerate()
|
||||
.for_each(|(chunk_idx, chunk)| {
|
||||
let base_idx = chunk_idx * BIGRAM_CHUNK_FILES;
|
||||
for (offset, file) in chunk.iter().enumerate() {
|
||||
let file_idx = base_idx + offset;
|
||||
|
||||
if file.is_binary() || file.size == 0 {
|
||||
return;
|
||||
}
|
||||
|
||||
READ_BUF.with(|read_cell| {
|
||||
let mut buf = read_cell.borrow_mut();
|
||||
let mut path_buf = [0u8; crate::simd_path::PATH_BUF_SIZE];
|
||||
|
||||
if let Some(content) = read_bigram_chunk(
|
||||
file,
|
||||
base_fd,
|
||||
base_path,
|
||||
arena,
|
||||
&mut buf[..],
|
||||
&mut path_buf,
|
||||
) {
|
||||
builder.add_file_content(&skip_builder, file_idx, content);
|
||||
}
|
||||
});
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
#[cfg(unix)]
|
||||
if base_fd >= 0 {
|
||||
unsafe { libc::close(base_fd) };
|
||||
}
|
||||
|
||||
let mut index = builder.compress(None);
|
||||
let skip_index = skip_builder.compress(Some(SKIP_INDEX_MIN_DENSITY_PCT));
|
||||
index.set_skip_index(skip_index);
|
||||
|
||||
// in progress bigram walk + rust's ignore crate allocates shit ton of garbage memory
|
||||
// all custom allocators would think this is available resource while we do not allocate
|
||||
// after the sync, so it's very important to let the unused memory go back to the OS
|
||||
crate::file_picker::hint_allocator_collect();
|
||||
|
||||
index
|
||||
}
|
||||
|
||||
/// Open the base directory for the `openat` fast path. Returns `-1` on
|
||||
/// failure — callers interpret a negative fd as "fall back to absolute
|
||||
/// paths".
|
||||
#[cfg(unix)]
|
||||
fn open_base_dir_fd(base_path: &std::path::Path) -> libc::c_int {
|
||||
use std::os::unix::ffi::OsStrExt;
|
||||
let mut cstr = [0u8; crate::simd_path::PATH_BUF_SIZE];
|
||||
let bytes = base_path.as_os_str().as_bytes();
|
||||
if bytes.len() >= cstr.len() {
|
||||
return -1;
|
||||
}
|
||||
cstr[..bytes.len()].copy_from_slice(bytes);
|
||||
// SAFETY: `cstr` is NUL-terminated by construction (zero-initialised,
|
||||
// and we only filled up to `bytes.len() < cstr.len()`).
|
||||
unsafe {
|
||||
libc::open(
|
||||
cstr.as_ptr() as *const std::os::raw::c_char,
|
||||
libc::O_RDONLY | libc::O_DIRECTORY,
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
/// Build a key the same way `add_file_content` does: two printable-ASCII
|
||||
/// bytes, lowercased, packed as `(hi << 8) | lo`.
|
||||
fn key(a: u8, b: u8) -> u16 {
|
||||
((a.to_ascii_lowercase() as u16) << 8) | b.to_ascii_lowercase() as u16
|
||||
}
|
||||
|
||||
/// Return the sorted list of (consec, skip) bigram keys that should appear
|
||||
/// for `content`. Used as the reference implementation.
|
||||
fn expected_bigrams(content: &[u8]) -> (Vec<u16>, Vec<u16>) {
|
||||
let mut consec: std::collections::BTreeSet<u16> = Default::default();
|
||||
let mut skip: std::collections::BTreeSet<u16> = Default::default();
|
||||
let printable = |b: u8| (32..=126).contains(&b);
|
||||
for i in 1..content.len() {
|
||||
let a = content[i - 1];
|
||||
let b = content[i];
|
||||
if printable(a) && printable(b) {
|
||||
consec.insert(key(a, b));
|
||||
}
|
||||
if i >= 2 {
|
||||
let a = content[i - 2];
|
||||
let b = content[i];
|
||||
if printable(a) && printable(b) {
|
||||
skip.insert(key(a, b));
|
||||
}
|
||||
}
|
||||
}
|
||||
(consec.into_iter().collect(), skip.into_iter().collect())
|
||||
}
|
||||
|
||||
/// Query: does the builder record file 0 as having this bigram set?
|
||||
fn builder_has_key_for_file_0(b: &BigramIndexBuilder, k: u16) -> bool {
|
||||
let col = b.lookup[k as usize].load(Ordering::Relaxed);
|
||||
if col == NO_COLUMN {
|
||||
return false;
|
||||
}
|
||||
b.column_bitset(col)[0] & 1 != 0
|
||||
}
|
||||
|
||||
fn run_and_compare(content: &[u8]) {
|
||||
let consec = BigramIndexBuilder::new(1);
|
||||
let skip = BigramIndexBuilder::new(1);
|
||||
consec.add_file_content(&skip, 0, content);
|
||||
|
||||
let (expected_consec, expected_skip) = expected_bigrams(content);
|
||||
|
||||
// Every expected bigram must be recorded.
|
||||
for k in &expected_consec {
|
||||
assert!(
|
||||
builder_has_key_for_file_0(&consec, *k),
|
||||
"consec bigram 0x{k:04x} missing for content {content:?}",
|
||||
);
|
||||
}
|
||||
for k in &expected_skip {
|
||||
assert!(
|
||||
builder_has_key_for_file_0(&skip, *k),
|
||||
"skip bigram 0x{k:04x} missing for content {content:?}",
|
||||
);
|
||||
}
|
||||
|
||||
// No unexpected bigrams — iterate lookup for set columns.
|
||||
for k in 0u32..=0xFFFF {
|
||||
let recorded_consec = builder_has_key_for_file_0(&consec, k as u16);
|
||||
let recorded_skip = builder_has_key_for_file_0(&skip, k as u16);
|
||||
if recorded_consec {
|
||||
assert!(
|
||||
expected_consec.contains(&(k as u16)),
|
||||
"unexpected consec bigram 0x{k:04x} in content {content:?}",
|
||||
);
|
||||
}
|
||||
if recorded_skip {
|
||||
assert!(
|
||||
expected_skip.contains(&(k as u16)),
|
||||
"unexpected skip bigram 0x{k:04x} in content {content:?}",
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn add_file_empty_is_noop() {
|
||||
let consec = BigramIndexBuilder::new(1);
|
||||
let skip = BigramIndexBuilder::new(1);
|
||||
consec.add_file_content(&skip, 0, b"");
|
||||
assert_eq!(consec.columns_used(), 0);
|
||||
assert_eq!(skip.columns_used(), 0);
|
||||
// populated counter not incremented for empty input
|
||||
assert_eq!(consec.populated.load(Ordering::Relaxed), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn add_file_single_byte_is_noop() {
|
||||
let consec = BigramIndexBuilder::new(1);
|
||||
let skip = BigramIndexBuilder::new(1);
|
||||
consec.add_file_content(&skip, 0, b"a");
|
||||
assert_eq!(consec.columns_used(), 0);
|
||||
assert_eq!(skip.columns_used(), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn add_file_two_bytes_consec_only() {
|
||||
// With exactly 2 bytes there's no skip bigram (needs i >= 2 in the loop).
|
||||
run_and_compare(b"ab");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn add_file_three_bytes_has_skip() {
|
||||
// "abc" -> consec {"ab", "bc"}, skip {"ac"}
|
||||
run_and_compare(b"abc");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn add_file_ascii_words() {
|
||||
run_and_compare(b"hello world");
|
||||
run_and_compare(b"the quick brown fox jumps over the lazy dog");
|
||||
run_and_compare(b"fn main() { println!(\"hi\"); }");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn add_file_case_is_lowered() {
|
||||
// Uppercase should be lowercased before keying, so "AB" == "ab".
|
||||
let upper = BigramIndexBuilder::new(1);
|
||||
let upper_skip = BigramIndexBuilder::new(1);
|
||||
upper.add_file_content(&upper_skip, 0, b"ABC");
|
||||
|
||||
let lower = BigramIndexBuilder::new(1);
|
||||
let lower_skip = BigramIndexBuilder::new(1);
|
||||
lower.add_file_content(&lower_skip, 0, b"abc");
|
||||
|
||||
// Both should have identical bigram keys.
|
||||
for k in 0u32..=0xFFFF {
|
||||
let u = builder_has_key_for_file_0(&upper, k as u16);
|
||||
let l = builder_has_key_for_file_0(&lower, k as u16);
|
||||
assert_eq!(u, l, "consec 0x{k:04x}: upper={u} lower={l}");
|
||||
let u = builder_has_key_for_file_0(&upper_skip, k as u16);
|
||||
let l = builder_has_key_for_file_0(&lower_skip, k as u16);
|
||||
assert_eq!(u, l, "skip 0x{k:04x}: upper={u} lower={l}");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn add_file_rejects_non_printable() {
|
||||
// Bigrams where either byte is outside 32..=126 are rejected. But
|
||||
// the skip-1 bigram can still connect two printable bytes across a
|
||||
// non-printable one: for "\0a\0b", consec sees no valid pair but
|
||||
// skip sees (a,b) at i=3. Use the reference implementation.
|
||||
run_and_compare(b"\0a\0b");
|
||||
|
||||
// All-zero input: truly nothing recorded.
|
||||
let consec = BigramIndexBuilder::new(1);
|
||||
let skip = BigramIndexBuilder::new(1);
|
||||
consec.add_file_content(&skip, 0, b"\0\0\0\0");
|
||||
assert_eq!(consec.columns_used(), 0);
|
||||
assert_eq!(skip.columns_used(), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn add_file_mixed_printable_and_control() {
|
||||
// "a\tb\nc d" — \t (9) and \n (10) are below 32. Consec:
|
||||
// (a, \t) x, (\t, b) x, (b, \n) x, (\n, c) x, (c, ' ') ok, (' ', d) ok
|
||||
// Skip (i-2, i):
|
||||
// (a, b) ok, (\t, \n) x, (b, c) ok, (\n, ' ') x, (c, d) ok
|
||||
run_and_compare(b"a\tb\nc d");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn add_file_repeats_are_deduped() {
|
||||
// "ababab" has many repeats of "ab", "ba" — each unique bigram should
|
||||
// be recorded exactly once (the stack-local `seen_*` dedup works).
|
||||
run_and_compare(b"ababababab");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn add_file_tombstone_separation() {
|
||||
// Two separate files share no bits; file 1's content doesn't bleed
|
||||
// into file 0's row and vice-versa.
|
||||
let consec = BigramIndexBuilder::new(2);
|
||||
let skip = BigramIndexBuilder::new(2);
|
||||
consec.add_file_content(&skip, 0, b"xy");
|
||||
consec.add_file_content(&skip, 1, b"zw");
|
||||
|
||||
let key_xy = key(b'x', b'y');
|
||||
let key_zw = key(b'z', b'w');
|
||||
|
||||
// file 0 has "xy" but not "zw"
|
||||
let col_xy = consec.lookup[key_xy as usize].load(Ordering::Relaxed);
|
||||
let col_zw = consec.lookup[key_zw as usize].load(Ordering::Relaxed);
|
||||
let bitset_xy = consec.column_bitset(col_xy)[0];
|
||||
let bitset_zw = consec.column_bitset(col_zw)[0];
|
||||
assert_eq!(bitset_xy & 0b01, 0b01, "file 0 should have xy");
|
||||
assert_eq!(bitset_zw & 0b01, 0, "file 0 should NOT have zw");
|
||||
assert_eq!(bitset_xy & 0b10, 0, "file 1 should NOT have xy");
|
||||
assert_eq!(bitset_zw & 0b10, 0b10, "file 1 should have zw");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn add_file_long_content() {
|
||||
// Stress test: ~8 KB of printable ASCII. Should complete without
|
||||
// overflowing any stack-local bitset and produce the full set.
|
||||
let mut buf = Vec::with_capacity(8192);
|
||||
for i in 0..8192 {
|
||||
buf.push(32u8 + ((i * 7) % 95) as u8); // cycle through printable range
|
||||
}
|
||||
run_and_compare(&buf);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn add_file_simd_and_scalar_agree() {
|
||||
// Cross-check: both code paths (scalar <128 bytes, SIMD ≥128) must
|
||||
// produce identical bigram sets for content that straddles the
|
||||
// threshold. Mix printable ASCII with some non-printable bytes and
|
||||
// repeats so the non-printable branch in the SIMD path exercises.
|
||||
let mut mixed = Vec::with_capacity(256);
|
||||
for i in 0..256usize {
|
||||
mixed.push(match i % 9 {
|
||||
0 => 0, // NUL
|
||||
1 => 0x7F, // DEL (just above 126)
|
||||
2 => b'\n', // below 32
|
||||
_ => 32 + ((i * 13) % 95) as u8,
|
||||
});
|
||||
}
|
||||
|
||||
run_and_compare(&mixed[..127]); // scalar path
|
||||
run_and_compare(&mixed); // SIMD path (256 bytes)
|
||||
run_and_compare(&mixed[..192]); // SIMD path with scalar tail
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn add_file_respects_file_count_boundary() {
|
||||
// file_count=100, file_idx=63 (last bit in word 0) and file_idx=64
|
||||
// (first bit in word 1). Make sure the word_idx math is right.
|
||||
let consec = BigramIndexBuilder::new(100);
|
||||
let skip = BigramIndexBuilder::new(100);
|
||||
consec.add_file_content(&skip, 63, b"ab");
|
||||
consec.add_file_content(&skip, 64, b"cd");
|
||||
|
||||
let kab = key(b'a', b'b');
|
||||
let kcd = key(b'c', b'd');
|
||||
let col_ab = consec.lookup[kab as usize].load(Ordering::Relaxed);
|
||||
let col_cd = consec.lookup[kcd as usize].load(Ordering::Relaxed);
|
||||
|
||||
let ab_bitset = consec.column_bitset(col_ab);
|
||||
let cd_bitset = consec.column_bitset(col_cd);
|
||||
// ab in word 0, bit 63
|
||||
assert_eq!(ab_bitset[0], 1u64 << 63);
|
||||
assert_eq!(ab_bitset[1], 0);
|
||||
// cd in word 1, bit 0
|
||||
assert_eq!(cd_bitset[0], 0);
|
||||
assert_eq!(cd_bitset[1], 1);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,998 @@
|
||||
//! Regex → bigram decomposition for the inverted bigram index.
|
||||
//!
|
||||
//! Parses a regex pattern with `regex-syntax`, walks the HIR to extract
|
||||
//! guaranteed bigram keys (u16), and evaluates them as an AND/OR query tree
|
||||
//! against [`BigramFilter`]'s inverted posting lists.
|
||||
//!
|
||||
//! Two bigram types are extracted:
|
||||
//! - **Consecutive** (gap=0): adjacent byte pairs `(pattern[i], pattern[i+1])`
|
||||
//! - **Sparse-1** (gap=1): pairs across a single-byte wildcard, e.g. `a.b → (a,b)`
|
||||
//!
|
||||
//! The sparse-1 extraction is the key feature: regex patterns like `foo.bar`
|
||||
//! yield the cross-boundary sparse-1 bigram `(o,b)` that provides strong
|
||||
//! filtering even when the `.` prevents any consecutive cross-boundary bigram.
|
||||
|
||||
use crate::bigram_filter::BigramFilter;
|
||||
use regex_syntax::hir::{Class, Hir, HirKind};
|
||||
use smallvec::SmallVec;
|
||||
use std::borrow::Cow;
|
||||
|
||||
/// Maximum byte values to enumerate from a character class.
|
||||
/// Larger classes are treated as unknown (no bigram extractable).
|
||||
const MAX_CLASS_EXPAND: usize = 16;
|
||||
|
||||
#[inline]
|
||||
fn consec_key(a: u8, b: u8) -> Option<u16> {
|
||||
let al = a.to_ascii_lowercase();
|
||||
let bl = b.to_ascii_lowercase();
|
||||
if (32..=126).contains(&al) && (32..=126).contains(&bl) {
|
||||
Some((al as u16) << 8 | bl as u16)
|
||||
} else {
|
||||
None
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone)]
|
||||
pub enum BigramQuery {
|
||||
Any,
|
||||
/// A consecutive bigram key to look up in the main index.
|
||||
Consec(u16),
|
||||
/// A skip-1 bigram key to look up in the skip sub-index.
|
||||
Skip1(u16),
|
||||
/// All children must match (intersect posting lists).
|
||||
And(Vec<BigramQuery>),
|
||||
/// At least one child must match (union posting lists).
|
||||
Or(Vec<BigramQuery>),
|
||||
}
|
||||
|
||||
/// SIMD-friendly bitwise OR of two equal-length bitsets.
|
||||
#[inline]
|
||||
fn bitset_or(a: &mut [u64], b: &[u64]) {
|
||||
a.iter_mut().zip(b.iter()).for_each(|(x, y)| *x |= *y);
|
||||
}
|
||||
|
||||
/// SIMD-friendly bitwise AND of two equal-length bitsets.
|
||||
#[inline]
|
||||
fn bitset_and(a: &mut [u64], b: &[u64]) {
|
||||
a.iter_mut().zip(b.iter()).for_each(|(x, y)| *x &= *y);
|
||||
}
|
||||
|
||||
impl BigramQuery {
|
||||
pub fn is_any(&self) -> bool {
|
||||
matches!(self, BigramQuery::Any)
|
||||
}
|
||||
|
||||
pub(crate) fn evaluate(&self, index: &BigramFilter) -> Option<Vec<u64>> {
|
||||
self.evaluate_cow(index).map(Cow::into_owned)
|
||||
}
|
||||
|
||||
fn evaluate_cow<'a>(&self, index: &'a BigramFilter) -> Option<Cow<'a, [u64]>> {
|
||||
match self {
|
||||
BigramQuery::Any => None,
|
||||
|
||||
BigramQuery::Consec(key) => {
|
||||
let col = index.lookup()[*key as usize];
|
||||
if col == u16::MAX {
|
||||
return None;
|
||||
}
|
||||
let words = index.words();
|
||||
let offset = col as usize * words;
|
||||
let data = index.dense_data();
|
||||
if offset + words > data.len() {
|
||||
return None;
|
||||
}
|
||||
Some(Cow::Borrowed(&data[offset..offset + words]))
|
||||
}
|
||||
|
||||
BigramQuery::Skip1(key) => {
|
||||
let skip = index.skip_index()?;
|
||||
let col = skip.lookup()[*key as usize];
|
||||
if col == u16::MAX {
|
||||
return None;
|
||||
}
|
||||
let words = skip.words();
|
||||
let offset = col as usize * words;
|
||||
let data = skip.dense_data();
|
||||
if offset + words > data.len() {
|
||||
return None;
|
||||
}
|
||||
Some(Cow::Borrowed(&data[offset..offset + words]))
|
||||
}
|
||||
|
||||
BigramQuery::And(children) => {
|
||||
let mut result: Option<Vec<u64>> = None;
|
||||
for child in children {
|
||||
if let Some(child_bits) = child.evaluate_cow(index) {
|
||||
result = Some(match result {
|
||||
None => child_bits.into_owned(),
|
||||
Some(mut r) => {
|
||||
bitset_and(&mut r, &child_bits);
|
||||
r
|
||||
}
|
||||
});
|
||||
}
|
||||
}
|
||||
result.map(Cow::Owned)
|
||||
}
|
||||
|
||||
BigramQuery::Or(children) => {
|
||||
if children.is_empty() {
|
||||
return None;
|
||||
}
|
||||
let mut result: Option<Vec<u64>> = None;
|
||||
for child in children {
|
||||
match child.evaluate_cow(index) {
|
||||
// Any branch can't be filtered → whole OR can't be filtered
|
||||
None => return None,
|
||||
Some(child_bits) => {
|
||||
result = Some(match result {
|
||||
None => child_bits.into_owned(),
|
||||
Some(mut r) => {
|
||||
bitset_or(&mut r, &child_bits);
|
||||
r
|
||||
}
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
result.map(Cow::Owned)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Intermediate state tracked during HIR traversal for bigram extraction.
|
||||
struct HirInfo {
|
||||
query: BigramQuery,
|
||||
/// Possible first bytes (lowercased, printable ASCII) when this node matches.
|
||||
first: Option<SmallVec<[u8; MAX_CLASS_EXPAND]>>,
|
||||
/// Possible last bytes.
|
||||
last: Option<SmallVec<[u8; MAX_CLASS_EXPAND]>>,
|
||||
/// Whether this node can match the empty string.
|
||||
can_be_empty: bool,
|
||||
}
|
||||
|
||||
impl HirInfo {
|
||||
fn empty() -> Self {
|
||||
Self {
|
||||
query: BigramQuery::Any,
|
||||
first: None,
|
||||
last: None,
|
||||
can_be_empty: true,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Prefilter fuzzy query. The algorithm is the following:
|
||||
/// we allow max_typos = min(len/3,2) every typo destroys at most 2 consecutive bigrams
|
||||
/// So out of N bigrams at least N - 2 * max_typos have to present in the matching fil
|
||||
pub(crate) fn fuzzy_to_bigram_query(query: &str, num_probes: usize) -> BigramQuery {
|
||||
let lower: Vec<u8> = query.bytes().map(|b| b.to_ascii_lowercase()).collect();
|
||||
|
||||
if lower.len() < 2 {
|
||||
return BigramQuery::Any;
|
||||
}
|
||||
|
||||
let max_typos = (lower.len() / 3).min(2);
|
||||
|
||||
// Extract all consecutive bigram keys.
|
||||
let bigram_keys: Vec<u16> = lower
|
||||
.windows(2)
|
||||
.filter_map(|w| consec_key(w[0], w[1]))
|
||||
.collect();
|
||||
|
||||
if bigram_keys.is_empty() {
|
||||
return BigramQuery::Any;
|
||||
}
|
||||
|
||||
// For very short queries (0 typos), AND all bigrams — exact subsequence.
|
||||
if max_typos == 0 {
|
||||
return simplify_and(
|
||||
bigram_keys
|
||||
.iter()
|
||||
.map(|&k| BigramQuery::Consec(k))
|
||||
.collect(),
|
||||
);
|
||||
}
|
||||
|
||||
// Pick evenly-spaced probe bigrams.
|
||||
let n = num_probes.min(bigram_keys.len());
|
||||
if n <= max_typos {
|
||||
// Too few probes to require anything useful.
|
||||
return simplify_or(
|
||||
bigram_keys
|
||||
.iter()
|
||||
.map(|&k| BigramQuery::Consec(k))
|
||||
.collect(),
|
||||
);
|
||||
}
|
||||
|
||||
let probes: Vec<u16> = if n == bigram_keys.len() {
|
||||
bigram_keys
|
||||
} else {
|
||||
(0..n)
|
||||
.map(|i| {
|
||||
let idx = i * (bigram_keys.len() - 1) / (n - 1);
|
||||
bigram_keys[idx]
|
||||
})
|
||||
.collect()
|
||||
};
|
||||
|
||||
let required = n - max_typos;
|
||||
|
||||
// If required == n, just AND all probes.
|
||||
if required >= n {
|
||||
return simplify_and(probes.iter().map(|&k| BigramQuery::Consec(k)).collect());
|
||||
}
|
||||
|
||||
// Generate all C(n, required) subsets → OR(AND(subset), ...)
|
||||
let mut branches = Vec::new();
|
||||
let mut combo = vec![0u16; required];
|
||||
combine(&probes, required, 0, 0, &mut combo, &mut branches);
|
||||
|
||||
simplify_or(branches)
|
||||
}
|
||||
|
||||
/// Build C(n, k) combination branches in-place on a fixed-size slice.
|
||||
fn combine(
|
||||
items: &[u16],
|
||||
k: usize,
|
||||
start: usize,
|
||||
depth: usize,
|
||||
combo: &mut [u16],
|
||||
branches: &mut Vec<BigramQuery>,
|
||||
) {
|
||||
if depth == k {
|
||||
branches.push(simplify_and(
|
||||
combo.iter().map(|&key| BigramQuery::Consec(key)).collect(),
|
||||
));
|
||||
return;
|
||||
}
|
||||
let remaining = k - depth;
|
||||
for i in start..=items.len() - remaining {
|
||||
combo[depth] = items[i];
|
||||
combine(items, k, i + 1, depth + 1, combo, branches);
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn regex_to_bigram_query(pattern: &str) -> BigramQuery {
|
||||
let mut parser = regex_syntax::ParserBuilder::new()
|
||||
.unicode(false)
|
||||
.utf8(false)
|
||||
.build();
|
||||
|
||||
let hir = match parser.parse(pattern) {
|
||||
Ok(h) => h,
|
||||
Err(_) => return BigramQuery::Any,
|
||||
};
|
||||
|
||||
decompose(&hir).query
|
||||
}
|
||||
|
||||
fn decompose(hir: &Hir) -> HirInfo {
|
||||
let can_be_empty = hir.properties().minimum_len().is_none_or(|n| n == 0);
|
||||
|
||||
match hir.kind() {
|
||||
HirKind::Empty => HirInfo::empty(),
|
||||
|
||||
HirKind::Literal(lit) => decompose_literal(lit.0.as_ref()),
|
||||
|
||||
HirKind::Class(class) => {
|
||||
let bytes = expand_class(class);
|
||||
match bytes {
|
||||
Some(b) if !b.is_empty() => HirInfo {
|
||||
query: BigramQuery::Any,
|
||||
first: Some(b.clone()),
|
||||
last: Some(b),
|
||||
can_be_empty,
|
||||
},
|
||||
_ => HirInfo {
|
||||
query: BigramQuery::Any,
|
||||
first: None,
|
||||
last: None,
|
||||
can_be_empty,
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
HirKind::Look(_) => HirInfo::empty(),
|
||||
|
||||
HirKind::Repetition(rep) => {
|
||||
let inner = decompose(&rep.sub);
|
||||
if rep.min == 0 {
|
||||
HirInfo {
|
||||
query: BigramQuery::Any,
|
||||
first: inner.first,
|
||||
last: inner.last,
|
||||
can_be_empty: true,
|
||||
}
|
||||
} else {
|
||||
// min >= 1: inner bigrams guaranteed
|
||||
let mut qs = Vec::new();
|
||||
if !inner.query.is_any() {
|
||||
qs.push(inner.query.clone());
|
||||
}
|
||||
// min >= 2: cross-boundary between consecutive occurrences
|
||||
if rep.min >= 2 {
|
||||
push_cross_consec(&mut qs, inner.last.as_deref(), inner.first.as_deref());
|
||||
}
|
||||
HirInfo {
|
||||
query: simplify_and(qs),
|
||||
first: inner.first,
|
||||
last: inner.last,
|
||||
can_be_empty,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
HirKind::Capture(cap) => decompose(&cap.sub),
|
||||
|
||||
HirKind::Concat(parts) => decompose_concat(parts),
|
||||
|
||||
HirKind::Alternation(alts) => decompose_alternation(alts),
|
||||
}
|
||||
}
|
||||
|
||||
/// Extract bigrams from a literal byte sequence.
|
||||
fn decompose_literal(bytes: &[u8]) -> HirInfo {
|
||||
if bytes.is_empty() {
|
||||
return HirInfo::empty();
|
||||
}
|
||||
|
||||
let lower: SmallVec<[u8; 64]> = bytes.iter().map(|b| b.to_ascii_lowercase()).collect();
|
||||
|
||||
if lower.len() == 1 {
|
||||
let b = lower[0];
|
||||
let first = if (32..=126).contains(&b) {
|
||||
Some(SmallVec::from_slice(&[b]))
|
||||
} else {
|
||||
None
|
||||
};
|
||||
return HirInfo {
|
||||
query: BigramQuery::Any,
|
||||
first: first.clone(),
|
||||
last: first,
|
||||
can_be_empty: false,
|
||||
};
|
||||
}
|
||||
|
||||
let mut qs: Vec<BigramQuery> = Vec::new();
|
||||
|
||||
// Consecutive bigrams
|
||||
for w in lower.windows(2) {
|
||||
if let Some(k) = consec_key(w[0], w[1]) {
|
||||
qs.push(BigramQuery::Consec(k));
|
||||
}
|
||||
}
|
||||
|
||||
// Skip-1 bigrams from the literal itself
|
||||
if lower.len() >= 3 {
|
||||
for i in 0..lower.len() - 2 {
|
||||
if let Some(k) = consec_key(lower[i], lower[i + 2]) {
|
||||
qs.push(BigramQuery::Skip1(k));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
let first_byte = lower[0];
|
||||
let last_byte = *lower.last().unwrap();
|
||||
|
||||
HirInfo {
|
||||
query: simplify_and(qs),
|
||||
first: if (32..=126).contains(&first_byte) {
|
||||
Some(SmallVec::from_slice(&[first_byte]))
|
||||
} else {
|
||||
None
|
||||
},
|
||||
last: if (32..=126).contains(&last_byte) {
|
||||
Some(SmallVec::from_slice(&[last_byte]))
|
||||
} else {
|
||||
None
|
||||
},
|
||||
can_be_empty: false,
|
||||
}
|
||||
}
|
||||
|
||||
fn decompose_concat(parts: &[Hir]) -> HirInfo {
|
||||
if parts.is_empty() {
|
||||
return HirInfo::empty();
|
||||
}
|
||||
|
||||
let infos: Vec<HirInfo> = parts.iter().map(decompose).collect();
|
||||
let mut qs: Vec<BigramQuery> = Vec::new();
|
||||
|
||||
// 1. Collect child bigrams
|
||||
for info in &infos {
|
||||
if !info.query.is_any() {
|
||||
qs.push(info.query.clone());
|
||||
}
|
||||
}
|
||||
|
||||
// 2. Dense cross-boundary between adjacent mandatory parts
|
||||
for pair in infos.windows(2) {
|
||||
if !pair[0].can_be_empty && !pair[1].can_be_empty {
|
||||
push_cross_consec(&mut qs, pair[0].last.as_deref(), pair[1].first.as_deref());
|
||||
}
|
||||
}
|
||||
|
||||
// 3. Sparse-1 cross-boundary: across a single 1-byte-wide middle part.
|
||||
// Catches `foo.bar` → sparse-1 `(o,b)` across the dot.
|
||||
if parts.len() >= 3 {
|
||||
for i in 0..parts.len() - 2 {
|
||||
let left = &infos[i];
|
||||
let mid = &parts[i + 1];
|
||||
let right = &infos[i + 2];
|
||||
|
||||
let min_len = mid.properties().minimum_len();
|
||||
let max_len = mid.properties().maximum_len();
|
||||
let is_1byte = min_len == Some(1) && max_len == Some(1);
|
||||
|
||||
if is_1byte && !left.can_be_empty && !right.can_be_empty {
|
||||
push_cross_skip1(&mut qs, left.last.as_deref(), right.first.as_deref());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
let first = collect_first(&infos);
|
||||
let last = collect_last(&infos);
|
||||
let can_be_empty = infos.iter().all(|i| i.can_be_empty);
|
||||
|
||||
HirInfo {
|
||||
query: simplify_and(qs),
|
||||
first,
|
||||
last,
|
||||
can_be_empty,
|
||||
}
|
||||
}
|
||||
|
||||
fn decompose_alternation(alts: &[Hir]) -> HirInfo {
|
||||
if alts.is_empty() {
|
||||
return HirInfo::empty();
|
||||
}
|
||||
|
||||
let infos: Vec<HirInfo> = alts.iter().map(decompose).collect();
|
||||
let query = simplify_or(infos.iter().map(|i| i.query.clone()).collect());
|
||||
let first = merge_byte_sets(infos.iter().map(|i| &i.first));
|
||||
let last = merge_byte_sets(infos.iter().map(|i| &i.last));
|
||||
let can_be_empty = infos.iter().any(|i| i.can_be_empty);
|
||||
|
||||
HirInfo {
|
||||
query,
|
||||
first,
|
||||
last,
|
||||
can_be_empty,
|
||||
}
|
||||
}
|
||||
|
||||
fn expand_class(class: &Class) -> Option<SmallVec<[u8; MAX_CLASS_EXPAND]>> {
|
||||
let mut bytes: SmallVec<[u8; MAX_CLASS_EXPAND]> = SmallVec::new();
|
||||
match class {
|
||||
Class::Bytes(bc) => {
|
||||
for range in bc.ranges() {
|
||||
let count = (range.end() as usize) - (range.start() as usize) + 1;
|
||||
if bytes.len() + count > MAX_CLASS_EXPAND {
|
||||
return None;
|
||||
}
|
||||
for b in range.start()..=range.end() {
|
||||
if (32..=126).contains(&b) {
|
||||
let lower = b.to_ascii_lowercase();
|
||||
if !bytes.contains(&lower) {
|
||||
bytes.push(lower);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
Class::Unicode(uc) => {
|
||||
for range in uc.ranges() {
|
||||
let start = range.start() as u32;
|
||||
let end = range.end() as u32;
|
||||
if start > 127 {
|
||||
continue;
|
||||
}
|
||||
let ascii_end = end.min(126) as u8;
|
||||
let ascii_start = start.max(32) as u8;
|
||||
if ascii_start > ascii_end {
|
||||
continue;
|
||||
}
|
||||
let count = (ascii_end - ascii_start) as usize + 1;
|
||||
if bytes.len() + count > MAX_CLASS_EXPAND {
|
||||
return None;
|
||||
}
|
||||
for b in ascii_start..=ascii_end {
|
||||
let lower = b.to_ascii_lowercase();
|
||||
if !bytes.contains(&lower) {
|
||||
bytes.push(lower);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
if bytes.is_empty() { None } else { Some(bytes) }
|
||||
}
|
||||
|
||||
/// Push consecutive cross-product bigrams into `qs`.
|
||||
fn push_cross_consec(qs: &mut Vec<BigramQuery>, last: Option<&[u8]>, first: Option<&[u8]>) {
|
||||
if let Some(q) = cross_product(last, first, false) {
|
||||
qs.push(q);
|
||||
}
|
||||
}
|
||||
|
||||
/// Push skip-1 cross-product bigrams into `qs`.
|
||||
fn push_cross_skip1(qs: &mut Vec<BigramQuery>, last: Option<&[u8]>, first: Option<&[u8]>) {
|
||||
if let Some(q) = cross_product(last, first, true) {
|
||||
qs.push(q);
|
||||
}
|
||||
}
|
||||
|
||||
fn cross_product(last: Option<&[u8]>, first: Option<&[u8]>, skip: bool) -> Option<BigramQuery> {
|
||||
let last = last?;
|
||||
let first = first?;
|
||||
let n = last.len() * first.len();
|
||||
if n == 0 || n > MAX_CLASS_EXPAND * MAX_CLASS_EXPAND {
|
||||
return None;
|
||||
}
|
||||
|
||||
let mut bigrams: Vec<BigramQuery> = Vec::with_capacity(n);
|
||||
for &l in last {
|
||||
for &f in first {
|
||||
if let Some(k) = consec_key(l, f) {
|
||||
let node = if skip {
|
||||
BigramQuery::Skip1(k)
|
||||
} else {
|
||||
BigramQuery::Consec(k)
|
||||
};
|
||||
bigrams.push(node);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
match bigrams.len() {
|
||||
0 => None,
|
||||
1 => Some(bigrams.into_iter().next().unwrap()),
|
||||
_ => Some(simplify_or(bigrams)),
|
||||
}
|
||||
}
|
||||
|
||||
fn collect_first(infos: &[HirInfo]) -> Option<SmallVec<[u8; MAX_CLASS_EXPAND]>> {
|
||||
let mut result: SmallVec<[u8; MAX_CLASS_EXPAND]> = SmallVec::new();
|
||||
for info in infos {
|
||||
if let Some(ref bytes) = info.first {
|
||||
for &b in bytes {
|
||||
if !result.contains(&b) {
|
||||
if result.len() >= MAX_CLASS_EXPAND {
|
||||
return None;
|
||||
}
|
||||
result.push(b);
|
||||
}
|
||||
}
|
||||
} else if !info.can_be_empty {
|
||||
return None;
|
||||
}
|
||||
if !info.can_be_empty {
|
||||
break;
|
||||
}
|
||||
}
|
||||
if result.is_empty() {
|
||||
None
|
||||
} else {
|
||||
Some(result)
|
||||
}
|
||||
}
|
||||
|
||||
fn collect_last(infos: &[HirInfo]) -> Option<SmallVec<[u8; MAX_CLASS_EXPAND]>> {
|
||||
let mut result: SmallVec<[u8; MAX_CLASS_EXPAND]> = SmallVec::new();
|
||||
for info in infos.iter().rev() {
|
||||
if let Some(ref bytes) = info.last {
|
||||
for &b in bytes {
|
||||
if !result.contains(&b) {
|
||||
if result.len() >= MAX_CLASS_EXPAND {
|
||||
return None;
|
||||
}
|
||||
result.push(b);
|
||||
}
|
||||
}
|
||||
} else if !info.can_be_empty {
|
||||
return None;
|
||||
}
|
||||
if !info.can_be_empty {
|
||||
break;
|
||||
}
|
||||
}
|
||||
if result.is_empty() {
|
||||
None
|
||||
} else {
|
||||
Some(result)
|
||||
}
|
||||
}
|
||||
|
||||
fn merge_byte_sets<'a>(
|
||||
iter: impl Iterator<Item = &'a Option<SmallVec<[u8; MAX_CLASS_EXPAND]>>>,
|
||||
) -> Option<SmallVec<[u8; MAX_CLASS_EXPAND]>> {
|
||||
let mut result: SmallVec<[u8; MAX_CLASS_EXPAND]> = SmallVec::new();
|
||||
for opt in iter {
|
||||
match opt {
|
||||
None => return None,
|
||||
Some(bytes) => {
|
||||
for &b in bytes {
|
||||
if !result.contains(&b) {
|
||||
if result.len() >= MAX_CLASS_EXPAND {
|
||||
return None;
|
||||
}
|
||||
result.push(b);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
if result.is_empty() {
|
||||
None
|
||||
} else {
|
||||
Some(result)
|
||||
}
|
||||
}
|
||||
|
||||
fn simplify_and(children: Vec<BigramQuery>) -> BigramQuery {
|
||||
let mut flat: Vec<BigramQuery> = Vec::new();
|
||||
for child in children {
|
||||
match child {
|
||||
BigramQuery::Any => {}
|
||||
BigramQuery::And(inner) => flat.extend(inner),
|
||||
other => flat.push(other),
|
||||
}
|
||||
}
|
||||
match flat.len() {
|
||||
0 => BigramQuery::Any,
|
||||
1 => flat.into_iter().next().unwrap(),
|
||||
_ => BigramQuery::And(flat),
|
||||
}
|
||||
}
|
||||
|
||||
fn simplify_or(children: Vec<BigramQuery>) -> BigramQuery {
|
||||
if children.iter().any(|c| c.is_any()) {
|
||||
return BigramQuery::Any;
|
||||
}
|
||||
let mut flat: Vec<BigramQuery> = Vec::new();
|
||||
for child in children {
|
||||
match child {
|
||||
BigramQuery::Or(inner) => flat.extend(inner),
|
||||
other => flat.push(other),
|
||||
}
|
||||
}
|
||||
match flat.len() {
|
||||
0 => BigramQuery::Any,
|
||||
1 => flat.into_iter().next().unwrap(),
|
||||
_ => BigramQuery::Or(flat),
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::bigram_filter::BigramIndexBuilder;
|
||||
|
||||
/// Build a tiny index from the given file contents for testing.
|
||||
fn build_test_index(files: &[&[u8]]) -> BigramFilter {
|
||||
let n = files.len();
|
||||
let consec_builder = BigramIndexBuilder::new(n);
|
||||
let skip_builder = BigramIndexBuilder::new(n);
|
||||
for (i, content) in files.iter().enumerate() {
|
||||
consec_builder.add_file_content(&skip_builder, i, content);
|
||||
}
|
||||
let mut idx = consec_builder.compress(Some(0));
|
||||
idx.set_skip_index(skip_builder.compress(Some(0)));
|
||||
idx
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn literal_pattern() {
|
||||
let idx = build_test_index(&[
|
||||
b"hello world", // 0: contains "hello"
|
||||
b"goodbye world", // 1: no "hello"
|
||||
b"say hello there", // 2: contains "hello"
|
||||
]);
|
||||
|
||||
let q = regex_to_bigram_query("hello");
|
||||
assert!(!q.is_any());
|
||||
|
||||
let candidates = q.evaluate(&idx).unwrap();
|
||||
assert!(BigramFilter::is_candidate(&candidates, 0));
|
||||
assert!(!BigramFilter::is_candidate(&candidates, 1));
|
||||
assert!(BigramFilter::is_candidate(&candidates, 2));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn alternation() {
|
||||
let idx = build_test_index(&[
|
||||
b"has foo in it", // 0
|
||||
b"has bar in it", // 1
|
||||
b"has xyz in it", // 2
|
||||
]);
|
||||
|
||||
let q = regex_to_bigram_query("foo|bar");
|
||||
assert!(!q.is_any());
|
||||
|
||||
let candidates = q.evaluate(&idx).unwrap();
|
||||
assert!(BigramFilter::is_candidate(&candidates, 0));
|
||||
assert!(BigramFilter::is_candidate(&candidates, 1));
|
||||
// xyz doesn't contain foo or bar bigrams
|
||||
assert!(!BigramFilter::is_candidate(&candidates, 2));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn wildcard_concat() {
|
||||
let idx = build_test_index(&[
|
||||
b"foo something bar", // 0
|
||||
b"foo only", // 1: has foo but not bar
|
||||
b"only bar", // 2: has bar but not foo
|
||||
]);
|
||||
|
||||
let q = regex_to_bigram_query("foo.*bar");
|
||||
assert!(!q.is_any());
|
||||
|
||||
let candidates = q.evaluate(&idx).unwrap();
|
||||
assert!(BigramFilter::is_candidate(&candidates, 0));
|
||||
// file 1 and 2 should be filtered (missing bigrams from "bar" / "foo")
|
||||
assert!(!BigramFilter::is_candidate(&candidates, 1));
|
||||
assert!(!BigramFilter::is_candidate(&candidates, 2));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn sparse1_across_dot() {
|
||||
// "a.b" should produce a skip-1 bigram (a,b)
|
||||
let idx = build_test_index(&[
|
||||
b"axb", // 0: has sparse-1 (a,b)
|
||||
b"ayb", // 1: has sparse-1 (a,b)
|
||||
b"xyz", // 2: no (a,b) at all
|
||||
]);
|
||||
|
||||
let q = regex_to_bigram_query("a.b");
|
||||
assert!(!q.is_any());
|
||||
|
||||
let candidates = q.evaluate(&idx).unwrap();
|
||||
assert!(BigramFilter::is_candidate(&candidates, 0));
|
||||
assert!(BigramFilter::is_candidate(&candidates, 1));
|
||||
assert!(!BigramFilter::is_candidate(&candidates, 2));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn sparse1_across_digit() {
|
||||
// "foo\dbar" → sparse-1 (o,b) across \d
|
||||
let idx = build_test_index(&[
|
||||
b"foo3bar baz", // 0: has all bigrams
|
||||
b"foobar baz", // 1: has consecutive (o,b) but pattern needs sparse-1
|
||||
b"xyz only", // 2: no relevant bigrams
|
||||
]);
|
||||
|
||||
let q = regex_to_bigram_query(r"foo\dbar");
|
||||
assert!(!q.is_any());
|
||||
|
||||
let candidates = q.evaluate(&idx).unwrap();
|
||||
assert!(BigramFilter::is_candidate(&candidates, 0));
|
||||
// file 1 may or may not match depending on what bigrams are in the index
|
||||
// (it has all the literal bigrams and also o,b as both consec and skip-1)
|
||||
// The important thing is file 2 is excluded:
|
||||
assert!(!BigramFilter::is_candidate(&candidates, 2));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn pure_wildcard_is_any() {
|
||||
let q = regex_to_bigram_query(".*");
|
||||
assert!(q.is_any());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn single_char_is_any() {
|
||||
let q = regex_to_bigram_query("a");
|
||||
assert!(q.is_any());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn invalid_regex_is_any() {
|
||||
let q = regex_to_bigram_query("[invalid");
|
||||
assert!(q.is_any());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn optional_group_excluded() {
|
||||
// (bar)? is optional — its bigrams are not required
|
||||
let q = regex_to_bigram_query("foo(bar)?baz");
|
||||
assert!(!q.is_any());
|
||||
|
||||
let idx = build_test_index(&[
|
||||
b"foobaz content", // 0: has foo+baz bigrams (bar absent)
|
||||
b"foobarbaz content", // 1: has everything
|
||||
b"xyz only", // 2: nothing
|
||||
]);
|
||||
|
||||
let candidates = q.evaluate(&idx).unwrap();
|
||||
assert!(BigramFilter::is_candidate(&candidates, 0));
|
||||
assert!(BigramFilter::is_candidate(&candidates, 1));
|
||||
assert!(!BigramFilter::is_candidate(&candidates, 2));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn repetition_min2_cross_boundary() {
|
||||
// (ab){2,} → bigram "ab" + cross-boundary "b","a"
|
||||
let q = regex_to_bigram_query("(ab){2,}");
|
||||
assert!(!q.is_any());
|
||||
|
||||
let idx = build_test_index(&[
|
||||
b"ababab", // 0: has "ab" and "b"->"a"
|
||||
b"abonly", // 1: has "ab" but not "b"->"a"
|
||||
b"xyz", // 2: nothing
|
||||
]);
|
||||
|
||||
let candidates = q.evaluate(&idx).unwrap();
|
||||
assert!(BigramFilter::is_candidate(&candidates, 0));
|
||||
assert!(!BigramFilter::is_candidate(&candidates, 2));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn two_dots_no_sparse1() {
|
||||
// "a..b" — two 1-byte parts between a and b, not a single 1-byte part
|
||||
// No sparse-1 (a,b) should be extracted
|
||||
let q = regex_to_bigram_query("a..b");
|
||||
// Single-char literals with 2 unknown bytes between → Any
|
||||
assert!(q.is_any());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn character_class_cross_boundary() {
|
||||
// [abc]de → cross-boundary OR(ad,bd,cd) + bigram de
|
||||
// All three class variants must appear in the corpus so the OR
|
||||
// branches are tracked in the index (untracked bigrams make the
|
||||
// OR conservatively return None, which is correct but untestable).
|
||||
let idx = build_test_index(&[
|
||||
b"ade content", // 0: has ad
|
||||
b"bde content", // 1: has bd
|
||||
b"cde content", // 2: has cd
|
||||
b"xde content", // 3: has de but not ad/bd/cd
|
||||
]);
|
||||
|
||||
let q = regex_to_bigram_query("[abc]de");
|
||||
assert!(!q.is_any());
|
||||
|
||||
let candidates = q.evaluate(&idx).unwrap();
|
||||
assert!(BigramFilter::is_candidate(&candidates, 0));
|
||||
assert!(BigramFilter::is_candidate(&candidates, 1));
|
||||
assert!(BigramFilter::is_candidate(&candidates, 2));
|
||||
// file 3 doesn't have ad/bd/cd so should be filtered
|
||||
assert!(!BigramFilter::is_candidate(&candidates, 3));
|
||||
}
|
||||
|
||||
// ── Helpers for inspecting query trees ──────────────────────────
|
||||
|
||||
fn has_consec(q: &BigramQuery, a: u8, b: u8) -> bool {
|
||||
let Some(key) = consec_key(a, b) else {
|
||||
return false;
|
||||
};
|
||||
match q {
|
||||
BigramQuery::Consec(k) => *k == key,
|
||||
BigramQuery::And(cs) | BigramQuery::Or(cs) => cs.iter().any(|c| has_consec(c, a, b)),
|
||||
_ => false,
|
||||
}
|
||||
}
|
||||
|
||||
fn has_skip1(q: &BigramQuery, a: u8, b: u8) -> bool {
|
||||
let Some(key) = consec_key(a, b) else {
|
||||
return false;
|
||||
};
|
||||
match q {
|
||||
BigramQuery::Skip1(k) => *k == key,
|
||||
BigramQuery::And(cs) | BigramQuery::Or(cs) => cs.iter().any(|c| has_skip1(c, a, b)),
|
||||
_ => false,
|
||||
}
|
||||
}
|
||||
|
||||
/// Bigram expectation: `("ab", is_skip1)`.
|
||||
/// The 2-char str is the byte pair; C = consecutive, S = skip-1.
|
||||
type Bg = (&'static str, bool);
|
||||
const C: bool = false;
|
||||
const S: bool = true;
|
||||
|
||||
/// Top 15+ commonly used regex patterns from
|
||||
/// https://digitalfortress.tech/tips/top-15-commonly-used-regex/
|
||||
/// plus typical grep patterns used by agentic tools.
|
||||
///
|
||||
/// Each entry: `(regex, Option<&[Bg]>)`.
|
||||
/// - `None` → pure classes / unsupported syntax, Any is acceptable.
|
||||
/// - `Some(&[..])` → must be non-Any, and every listed bigram must appear.
|
||||
#[test]
|
||||
fn common_regex_patterns() {
|
||||
#[rustfmt::skip]
|
||||
let cases: &[(&str, Option<&[Bg]>)] = &[
|
||||
// ── Pure-class / anchor / unsupported → Any is fine ──────
|
||||
(r"^\d+$", None), // 1. whole numbers
|
||||
(r"^\d*\.\d+$", None), // 2. decimals
|
||||
(r"^\d*(\.\d+)?$", None), // 3. whole + decimal
|
||||
(r"^-?\d*(\.\d+)?$", None), // 4. neg/pos decimal
|
||||
(r"[-]?[0-9]+[,.]?[0-9]*([/][0-9]+[,.]?[0-9]*)*", None), // 5. fractions
|
||||
(r"^[a-zA-Z0-9]*$", None), // 6. alphanumeric
|
||||
(r"^[a-zA-Z0-9 ]*$", None), // 7. alphanum + space
|
||||
(r"^([a-zA-Z0-9._%-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,6})*$", None), // 8. email
|
||||
(r"^([a-z0-9_\.\+-]+)@([\da-z\.-]+)\.([a-z\.]{2,6})$", None), // 9. email v2
|
||||
(r"(?=(.*[0-9]))(?=.*[!@#$%^&*()\[\]{}\-_+=~`|:;<>,./?\x5c])(?=.*[a-z])(?=(.*[A-Z]))(?=(.*)).{8,}", None), // 10. complex pw
|
||||
(r"(?=(.*[0-9]))((?=.*[A-Za-z0-9])(?=.*[A-Z])(?=.*[a-z]))^.{8,}$", None), // 11. moderate pw
|
||||
(r"^[a-z0-9_-]{3,16}$", None), // 12. username
|
||||
(r"(https?://)?(www\.)?[-a-zA-Z0-9@:%._\+~#=]{2,256}\.[a-z]{2,6}\b([-a-zA-Z0-9@:%_\+.~#?&//=]*)", None), // 14. URL optional
|
||||
(r"^(([0-9]|[1-9][0-9]|1[0-9]{2}|2[0-4][0-9]|25[0-5])\.){3}([0-9]|[1-9][0-9]|1[0-9]{2}|2[0-4][0-9]|25[0-5])$", None), // 15. IPv4
|
||||
(r"(([0-9a-fA-F]{1,4}:){7,7}[0-9a-fA-F]{1,4}|([0-9a-fA-F]{1,4}:){1,7}:|([0-9a-fA-F]{1,4}:){1,6}:[0-9a-fA-F]{1,4}|([0-9a-fA-F]{1,4}:){1,5}(:[0-9a-fA-F]{1,4}){1,2}|([0-9a-fA-F]{1,4}:){1,4}(:[0-9a-fA-F]{1,4}){1,3}|([0-9a-fA-F]{1,4}:){1,3}(:[0-9a-fA-F]{1,4}){1,4}|([0-9a-fA-F]{1,4}:){1,2}(:[0-9a-fA-F]{1,4}){1,5}|[0-9a-fA-F]{1,4}:((:[0-9a-fA-F]{1,4}){1,6})|:((:[0-9a-fA-F]{1,4}){1,7}|:)|fe80:(:[0-9a-fA-F]{0,4}){0,4}%[0-9a-zA-Z]{1,}|::(ffff(:0{1,4}){0,1}:){0,1}((25[0-5]|(2[0-4]|1{0,1}[0-9]){0,1}[0-9])\.){3,3}(25[0-5]|(2[0-4]|1{0,1}[0-9]){0,1}[0-9])|([0-9a-fA-F]{1,4}:){1,4}:((25[0-5]|(2[0-4]|1{0,1}[0-9]){0,1}[0-9])\.){3,3}(25[0-5]|(2[0-4]|1{0,1}[0-9]){0,1}[0-9]))", None), // 16. IPv6
|
||||
(r"[12]\d{3}-(0[1-9]|1[0-2])-(0[1-9]|[12]\d|3[01])", None), // 17. date
|
||||
(r"^(0?[1-9]|1[0-2]):[0-5][0-9]$", None), // 18. time 12h
|
||||
(r"((1[0-2]|0?[1-9]):([0-5][0-9]) ?([AaPp][Mm]))", None), // 19. time AM/PM
|
||||
(r"^(0[0-9]|1[0-9]|2[0-3]):[0-5][0-9]$", None), // 20. time 24h
|
||||
(r"^([0-9]|0[0-9]|1[0-9]|2[0-3]):[0-5][0-9]$", None), // 21. time 24h v2
|
||||
(r"(?:[01]\d|2[0123]):(?:[012345]\d):(?:[012345]\d)", None), // 22. time+sec
|
||||
(r"</?[\w\s]*>|<.+[\W]>", None), // 23. HTML tag
|
||||
(r"\bon\w+=\S+(?=.*>)", None), // 24. inline JS
|
||||
(r"^[a-z0-9]+(?:-[a-z0-9]+)*$", None), // 25. slug
|
||||
(r"(\b\w+\b)(?=.*\b\1\b)", None), // 26. dup words
|
||||
(r"^[\w,\s-]+\.[A-Za-z]{3}$", None), // 27. filename
|
||||
(r"^[A-PR-WY][1-9]\d\s?\d{4}[1-9]$", None), // 28. HK ID
|
||||
|
||||
// ── Patterns with extractable literal bigrams ────────────
|
||||
|
||||
// 13. URL with required protocol
|
||||
(r"https?://(www\.)?[-a-zA-Z0-9@:%._\+~#=]{2,256}\.[a-z]{2,6}\b([-a-zA-Z0-9@:%_\+.~#?&//=]*)", Some(&[
|
||||
("ht", C), ("tt", C), ("tp", C), // from "http"
|
||||
("ht", S), ("tp", S), // from "http" skip-1
|
||||
(":/", C), ("//", C), // from "://"
|
||||
])),
|
||||
|
||||
// 29. fn\s+\w+
|
||||
(r"fn\s+\w+", Some(&[
|
||||
("fn", C), // from "fn"
|
||||
("n ", C), // cross-boundary: 'n' → \s starts ' '
|
||||
])),
|
||||
|
||||
// 30. use\s+crate::
|
||||
(r"use\s+crate::", Some(&[
|
||||
("us", C), ("se", C), ("ue", S), // from "use"
|
||||
("cr", C), ("ra", C), ("at", C), // from "crate"
|
||||
("te", C), ("::", C),
|
||||
("ca", S), ("rt", S), ("ae", S), // "crate" skip-1
|
||||
])),
|
||||
|
||||
// 31. unwrap\(\)|expect\(
|
||||
(r"unwrap\(\)|expect\(", Some(&[
|
||||
("nw", C), ("wr", C), ("ra", C), // "unwrap("
|
||||
("ap", C), ("p(", C),
|
||||
("xp", C), ("pe", C), ("ec", C), // "expect("
|
||||
("ct", C), ("t(", C),
|
||||
])),
|
||||
|
||||
// 32. TODO|FIXME|HACK
|
||||
(r"TODO|FIXME|HACK", Some(&[
|
||||
("to", C), ("od", C), ("do", C), // "TODO"
|
||||
("fi", C), ("ix", C), ("xm", C), // "FIXME"
|
||||
("me", C),
|
||||
("ha", C), ("ac", C), ("ck", C), // "HACK"
|
||||
("hc", S), ("ak", S), // "HACK" skip-1
|
||||
])),
|
||||
];
|
||||
|
||||
for (i, &(pattern, expected)) in cases.iter().enumerate() {
|
||||
let q = regex_to_bigram_query(pattern);
|
||||
|
||||
if let Some(bigrams) = expected {
|
||||
assert!(
|
||||
!q.is_any(),
|
||||
"#{i} {pattern:?}: expected bigrams but got Any"
|
||||
);
|
||||
|
||||
for &(pair, skip) in bigrams {
|
||||
let b = pair.as_bytes();
|
||||
debug_assert_eq!(b.len(), 2, "bigram must be 2 chars: {pair:?}");
|
||||
let found = if skip {
|
||||
has_skip1(&q, b[0], b[1])
|
||||
} else {
|
||||
has_consec(&q, b[0], b[1])
|
||||
};
|
||||
let kind = if skip { "skip-1" } else { "consec" };
|
||||
assert!(found, "#{i} {pattern:?}: missing {kind} bigram {pair:?}");
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,662 @@
|
||||
//! SIMD-accelerated case-insensitive substring search.
|
||||
//!
|
||||
//! Implementations (fastest → simplest):
|
||||
//! - `search_packed_pair`: AVX2 packed-pair scan (two rare bytes at known offsets)
|
||||
//! - `search`: memchr2 first-byte scan + verify
|
||||
//!
|
||||
//! The packed-pair approach mirrors what `memchr::memmem` does internally for
|
||||
//! case-sensitive search — pick two rare bytes from the needle, SIMD-scan for
|
||||
//! both simultaneously, verify candidates. This gives quadratic selectivity
|
||||
//! over the single-byte memchr2 approach.
|
||||
|
||||
// this is stolen from the memchr2 crate
|
||||
const BYTE_FREQUENCIES: [u8; 256] = [
|
||||
55, 52, 51, 50, 49, 48, 47, 46, 45, 103, 242, 66, 67, 229, 44, 43, // 0x00
|
||||
42, 41, 40, 39, 38, 37, 36, 35, 34, 33, 56, 32, 31, 30, 29, 28, // 0x10
|
||||
255, 148, 164, 149, 136, 160, 155, 173, 221, 222, 134, 122, 232, 202, 215, 224, // 0x20
|
||||
208, 220, 204, 187, 183, 179, 177, 168, 178, 200, 226, 195, 154, 184, 174, 126, // 0x30
|
||||
120, 191, 157, 194, 170, 189, 162, 161, 150, 193, 142, 137, 171, 176, 185,
|
||||
167, // 0x40 A-O
|
||||
186, 112, 175, 192, 188, 156, 140, 143, 123, 133, 128, 147, 138, 146, 114,
|
||||
223, // 0x50 P-_
|
||||
151, 249, 216, 238, 236, 253, 227, 218, 230, 247, 135, 180, 241, 233, 246,
|
||||
244, // 0x60 a-o
|
||||
231, 139, 245, 243, 251, 235, 201, 196, 240, 214, 152, 182, 205, 181, 127,
|
||||
27, // 0x70 p-DEL
|
||||
212, 211, 210, 213, 228, 197, 169, 159, 131, 172, 105, 80, 98, 96, 97, 81, // 0x80
|
||||
207, 145, 116, 115, 144, 130, 153, 121, 107, 132, 109, 110, 124, 111, 82, 108, // 0x90
|
||||
118, 141, 113, 129, 119, 125, 165, 117, 92, 106, 83, 72, 99, 93, 65, 79, // 0xa0
|
||||
166, 237, 163, 199, 190, 225, 209, 203, 198, 217, 219, 206, 234, 248, 158, 239, // 0xb0
|
||||
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, // 0xc0
|
||||
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, // 0xd0
|
||||
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, // 0xe0
|
||||
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, // 0xf0
|
||||
];
|
||||
|
||||
#[inline]
|
||||
fn ascii_fold_byte(b: u8) -> u8 {
|
||||
if b.is_ascii_uppercase() { b | 0x20 } else { b }
|
||||
}
|
||||
|
||||
/// Toggle ASCII letter case by flipping bit 5.
|
||||
/// `'n' → 'N'`, `'N' → 'n'`.
|
||||
#[inline]
|
||||
fn ascii_swap_case(b: u8) -> u8 {
|
||||
b ^ 0x20
|
||||
}
|
||||
|
||||
/// Effective frequency rank for a case-insensitive byte position.
|
||||
/// Takes the max of lower/upper ranks because we must scan for both.
|
||||
#[inline]
|
||||
fn case_insensitive_rank(lower: u8) -> u8 {
|
||||
if lower.is_ascii_lowercase() {
|
||||
let upper = ascii_swap_case(lower);
|
||||
BYTE_FREQUENCIES[lower as usize].max(BYTE_FREQUENCIES[upper as usize])
|
||||
} else {
|
||||
BYTE_FREQUENCIES[lower as usize]
|
||||
}
|
||||
}
|
||||
|
||||
/// Pick two needle positions with the rarest bytes (case-insensitive).
|
||||
/// Returns (index1, index2) where index1 <= index2.
|
||||
fn select_rare_pair(needle_lower: &[u8]) -> (usize, usize) {
|
||||
debug_assert!(needle_lower.len() >= 2);
|
||||
|
||||
let mut best1 = (u8::MAX, 0usize); // (rank, position)
|
||||
let mut best2 = (u8::MAX, 1usize);
|
||||
|
||||
for (i, &b) in needle_lower.iter().enumerate() {
|
||||
let r = case_insensitive_rank(b);
|
||||
if r < best1.0 {
|
||||
best2 = best1;
|
||||
best1 = (r, i);
|
||||
} else if r < best2.0 && i != best1.1 {
|
||||
best2 = (r, i);
|
||||
}
|
||||
}
|
||||
|
||||
let i1 = best1.1.min(best2.1);
|
||||
let i2 = best1.1.max(best2.1);
|
||||
(i1, i2)
|
||||
}
|
||||
|
||||
#[inline]
|
||||
fn verify_scalar(h: *const u8, needle_lower: &[u8]) -> bool {
|
||||
for (i, _) in needle_lower.iter().enumerate() {
|
||||
if ascii_fold_byte(unsafe { *h.add(i) }) != needle_lower[i] {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
true
|
||||
}
|
||||
|
||||
/// AVX2 case-insensitive verify: checks whether `needle_lower` matches
|
||||
/// the haystack bytes starting at `h`, treating ASCII uppercase as lowercase.
|
||||
///
|
||||
/// Processes 32 bytes at a time using a SIMD trick: AVX2 only has a
|
||||
/// **signed** byte compare (`cmpgt`), but we need an **unsigned** range
|
||||
/// check (`'A' <= byte <= 'Z'`). The trick is to XOR every byte with
|
||||
/// `0x80`, which maps the unsigned range `[0, 255]` into the signed range
|
||||
/// `[-128, 127]` while preserving order. After the flip, signed `cmpgt`
|
||||
/// gives correct unsigned comparisons.
|
||||
///
|
||||
/// Once we know which bytes are uppercase, we set bit 5 (`0x20`) on them
|
||||
/// — this converts `'A'..'Z'` to `'a'..'z'` — then compare against the
|
||||
/// pre-lowered needle.
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
#[target_feature(enable = "avx2")]
|
||||
unsafe fn verify_avx2(h: *const u8, needle_lower: &[u8]) -> bool {
|
||||
use core::arch::x86_64::*;
|
||||
|
||||
let len = needle_lower.len();
|
||||
let mut i = 0usize;
|
||||
|
||||
// Broadcast constants used every iteration:
|
||||
//
|
||||
// flip = 0x80 in every lane — XOR converts unsigned→signed domain
|
||||
// a_minus_1 = ('A' - 1) ^ 0x80 — lower bound for the range check (signed)
|
||||
// z_plus_1 = ('Z' + 1) ^ 0x80 — upper bound for the range check (signed)
|
||||
// bit20 = 0x20 in every lane — OR this onto uppercase bytes to lowercase them
|
||||
let flip = _mm256_set1_epi8(0x80u8 as i8);
|
||||
let a_minus_1 = _mm256_set1_epi8((b'A' - 1) as i8 ^ 0x80u8 as i8);
|
||||
let z_plus_1 = _mm256_set1_epi8((b'Z' + 1) as i8 ^ 0x80u8 as i8);
|
||||
let bit20 = _mm256_set1_epi8(0x20u8 as i8);
|
||||
|
||||
while i + 32 <= len {
|
||||
// Load 32 bytes from the haystack candidate position.
|
||||
let hv = unsafe { _mm256_loadu_si256(h.add(i) as *const __m256i) };
|
||||
// Load 32 bytes from the pre-lowercased needle.
|
||||
let nv = unsafe { _mm256_loadu_si256(needle_lower.as_ptr().add(i) as *const __m256i) };
|
||||
|
||||
// Flip into signed domain: x = hv ^ 0x80.
|
||||
// After this, unsigned ordering is preserved under signed compare.
|
||||
let x = _mm256_xor_si256(hv, flip);
|
||||
|
||||
// ge_a[lane] = 0xFF if x[lane] > a_minus_1, i.e. hv[lane] >= 'A' (unsigned).
|
||||
let ge_a = _mm256_cmpgt_epi8(x, a_minus_1);
|
||||
// le_z[lane] = 0xFF if z_plus_1 > x[lane], i.e. hv[lane] <= 'Z' (unsigned).
|
||||
let le_z = _mm256_cmpgt_epi8(z_plus_1, x);
|
||||
// upper[lane] = 0xFF only for bytes in the range 'A'..='Z'.
|
||||
let upper = _mm256_and_si256(ge_a, le_z);
|
||||
|
||||
// Case-fold: set bit 5 on uppercase bytes → converts 'A'..'Z' to 'a'..'z'.
|
||||
// Non-letter bytes are untouched because their `upper` lane is 0x00.
|
||||
let folded = _mm256_or_si256(hv, _mm256_and_si256(upper, bit20));
|
||||
|
||||
// Compare the folded haystack against the lowercase needle.
|
||||
let eq = _mm256_cmpeq_epi8(folded, nv);
|
||||
// movemask extracts the high bit of each lane into a 32-bit mask.
|
||||
// All-equal → all high bits set → mask == 0xFFFFFFFF == -1i32.
|
||||
if _mm256_movemask_epi8(eq) != -1i32 {
|
||||
return false;
|
||||
}
|
||||
|
||||
i += 32;
|
||||
}
|
||||
|
||||
// Scalar tail: handle remaining bytes that don't fill a full 32-byte vector.
|
||||
while i < len {
|
||||
if ascii_fold_byte(unsafe { *h.add(i) }) != needle_lower[i] {
|
||||
return false;
|
||||
}
|
||||
i += 1;
|
||||
}
|
||||
true
|
||||
}
|
||||
|
||||
// ======== NEON + dotprod (aarch64) ===========================================
|
||||
|
||||
/// Extract a 16-bit bitmask from a NEON comparison result (each byte 0x00 or 0xFF).
|
||||
/// Bit *i* of the result corresponds to byte *i* of the input vector.
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
#[target_feature(enable = "neon")]
|
||||
#[inline]
|
||||
unsafe fn neon_movemask(v: core::arch::aarch64::uint8x16_t) -> u16 {
|
||||
use core::arch::aarch64::*;
|
||||
|
||||
// AND each byte with its bit-position mask, then horizontally sum each half.
|
||||
// Max possible sum per half = 1+2+4+8+16+32+64+128 = 255, fits in u8.
|
||||
static BITS: [u8; 16] = [1, 2, 4, 8, 16, 32, 64, 128, 1, 2, 4, 8, 16, 32, 64, 128];
|
||||
let bit_mask = unsafe { vld1q_u8(BITS.as_ptr()) };
|
||||
let masked = vandq_u8(v, bit_mask);
|
||||
let lo = vaddv_u8(vget_low_u8(masked));
|
||||
let hi = vaddv_u8(vget_high_u8(masked));
|
||||
(lo as u16) | ((hi as u16) << 8)
|
||||
}
|
||||
|
||||
/// NEON + dotprod case-insensitive verify.
|
||||
///
|
||||
/// Uses unsigned range checks (NEON has `vcge`/`vcle` for unsigned bytes
|
||||
/// no XOR-0x80 trick needed unlike AVX2) to detect uppercase ASCII, folds
|
||||
/// to lowercase, then checks equality via UDOT: XOR the folded haystack
|
||||
/// with the pre-lowered needle and dot-product the difference with itself.
|
||||
/// Any non-zero byte produces a non-zero u32 lane.
|
||||
///
|
||||
/// The UDOT instruction is emitted via inline asm because the `vdotq_u32`
|
||||
/// intrinsic is still behind an unstable feature gate on stable Rust.
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
#[target_feature(enable = "neon,dotprod")]
|
||||
unsafe fn verify_neon_dotprod(h: *const u8, needle_lower: &[u8]) -> bool {
|
||||
use core::arch::aarch64::*;
|
||||
|
||||
let len = needle_lower.len();
|
||||
let mut i = 0usize;
|
||||
|
||||
let a_val = vdupq_n_u8(b'A');
|
||||
let z_val = vdupq_n_u8(b'Z');
|
||||
let bit20 = vdupq_n_u8(0x20);
|
||||
|
||||
while i + 16 <= len {
|
||||
let hv = unsafe { vld1q_u8(h.add(i)) };
|
||||
let nv = unsafe { vld1q_u8(needle_lower.as_ptr().add(i)) };
|
||||
|
||||
// Unsigned range check: 'A' <= byte <= 'Z'
|
||||
let upper = vandq_u8(vcgeq_u8(hv, a_val), vcleq_u8(hv, z_val));
|
||||
// Case-fold: set bit 5 on uppercase bytes → 'A'..'Z' → 'a'..'z'
|
||||
let folded = vorrq_u8(hv, vandq_u8(upper, bit20));
|
||||
|
||||
// XOR with needle — all-zero iff every byte matches.
|
||||
let xored = veorq_u8(folded, nv);
|
||||
|
||||
// UDOT: dot(xored, xored) sums squares of 4 consecutive byte
|
||||
// differences into each of the 4 u32 lanes (accumulates into zero).
|
||||
// Any non-zero byte produces a positive u32 contribution.
|
||||
let dots: uint32x4_t;
|
||||
let zero = vdupq_n_u32(0);
|
||||
unsafe {
|
||||
core::arch::asm!(
|
||||
"udot {d:v}.4s, {a:v}.16b, {b:v}.16b",
|
||||
d = inlateout(vreg) zero => dots,
|
||||
a = in(vreg) xored,
|
||||
b = in(vreg) xored,
|
||||
);
|
||||
}
|
||||
|
||||
if vmaxvq_u32(dots) != 0 {
|
||||
return false;
|
||||
}
|
||||
|
||||
i += 16;
|
||||
}
|
||||
|
||||
// Scalar tail
|
||||
while i < len {
|
||||
if ascii_fold_byte(unsafe { *h.add(i) }) != needle_lower[i] {
|
||||
return false;
|
||||
}
|
||||
i += 1;
|
||||
}
|
||||
true
|
||||
}
|
||||
|
||||
/// NEON packed-pair kernel: scan 16 haystack positions per iteration,
|
||||
/// checking two rare bytes (case-insensitive) simultaneously.
|
||||
/// Same algorithm as the AVX2 version but with 128-bit vectors.
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
#[target_feature(enable = "neon")]
|
||||
unsafe fn search_packed_pair_neon(
|
||||
haystack: &[u8],
|
||||
needle_lower: &[u8],
|
||||
i1: usize,
|
||||
i2: usize,
|
||||
) -> bool {
|
||||
use core::arch::aarch64::*;
|
||||
|
||||
let n = needle_lower.len();
|
||||
let hlen = haystack.len();
|
||||
let ptr = haystack.as_ptr();
|
||||
let last_start = hlen - n;
|
||||
|
||||
let b1 = needle_lower[i1];
|
||||
let b1_alt = if b1.is_ascii_lowercase() {
|
||||
ascii_swap_case(b1)
|
||||
} else {
|
||||
b1
|
||||
};
|
||||
let b2 = needle_lower[i2];
|
||||
let b2_alt = if b2.is_ascii_lowercase() {
|
||||
ascii_swap_case(b2)
|
||||
} else {
|
||||
b2
|
||||
};
|
||||
|
||||
let v1_lo = vdupq_n_u8(b1);
|
||||
let v1_hi = vdupq_n_u8(b1_alt);
|
||||
let v2_lo = vdupq_n_u8(b2);
|
||||
let v2_hi = vdupq_n_u8(b2_alt);
|
||||
|
||||
let max_idx = i1.max(i2);
|
||||
let max_offset = hlen.saturating_sub(max_idx + 16);
|
||||
let mut offset = 0usize;
|
||||
|
||||
while offset <= max_offset {
|
||||
let chunk1 = unsafe { vld1q_u8(ptr.add(offset + i1)) };
|
||||
let chunk2 = unsafe { vld1q_u8(ptr.add(offset + i2)) };
|
||||
|
||||
// Case-insensitive match: OR both case variants, then AND the two positions.
|
||||
let eq1 = vorrq_u8(vceqq_u8(chunk1, v1_lo), vceqq_u8(chunk1, v1_hi));
|
||||
let eq2 = vorrq_u8(vceqq_u8(chunk2, v2_lo), vceqq_u8(chunk2, v2_hi));
|
||||
|
||||
let mut mask = unsafe { neon_movemask(vandq_u8(eq1, eq2)) };
|
||||
|
||||
while mask != 0 {
|
||||
let bit = mask.trailing_zeros() as usize;
|
||||
let candidate = offset + bit;
|
||||
if candidate > last_start {
|
||||
return false;
|
||||
}
|
||||
if unsafe { verify_dispatch(ptr.add(candidate), needle_lower) } {
|
||||
return true;
|
||||
}
|
||||
mask &= mask - 1;
|
||||
}
|
||||
|
||||
offset += 16;
|
||||
}
|
||||
|
||||
// Tail: remaining positions that couldn't fill a full vector.
|
||||
if offset <= last_start {
|
||||
let rare_pos =
|
||||
if case_insensitive_rank(needle_lower[i1]) <= case_insensitive_rank(needle_lower[i2]) {
|
||||
i1
|
||||
} else {
|
||||
i2
|
||||
};
|
||||
let rare_byte = needle_lower[rare_pos];
|
||||
let tail_start = offset + rare_pos;
|
||||
let tail_end = last_start + rare_pos + 1;
|
||||
if tail_start < tail_end {
|
||||
let tail_space = &haystack[tail_start..tail_end];
|
||||
if rare_byte.is_ascii_lowercase() {
|
||||
for pos in memchr::memchr2_iter(rare_byte, ascii_swap_case(rare_byte), tail_space) {
|
||||
let candidate = offset + pos;
|
||||
if unsafe { verify_dispatch(ptr.add(candidate), needle_lower) } {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
} else {
|
||||
for pos in memchr::memchr_iter(rare_byte, tail_space) {
|
||||
let candidate = offset + pos;
|
||||
if unsafe { verify_dispatch(ptr.add(candidate), needle_lower) } {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
false
|
||||
}
|
||||
|
||||
#[inline]
|
||||
unsafe fn verify_dispatch(h: *const u8, needle_lower: &[u8]) -> bool {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
{
|
||||
if needle_lower.len() >= 32 && std::is_x86_feature_detected!("avx2") {
|
||||
return unsafe { verify_avx2(h, needle_lower) };
|
||||
}
|
||||
}
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
{
|
||||
if needle_lower.len() >= 16 && std::arch::is_aarch64_feature_detected!("dotprod") {
|
||||
return unsafe { verify_neon_dotprod(h, needle_lower) };
|
||||
}
|
||||
}
|
||||
|
||||
verify_scalar(h, needle_lower)
|
||||
}
|
||||
|
||||
// ── Packed-pair search (AVX2) ───────────────────────────────────────────
|
||||
|
||||
/// AVX2 packed-pair kernel: scan 32 haystack positions per iteration,
|
||||
/// checking two rare bytes (case-insensitive) simultaneously.
|
||||
/// 4 cmpeq + 2 or + 1 and + 1 movemask per 32 bytes — same memory
|
||||
/// bandwidth as memchr2 but quadratic selectivity.
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
#[target_feature(enable = "avx2")]
|
||||
unsafe fn search_packed_pair_avx2(
|
||||
haystack: &[u8],
|
||||
needle_lower: &[u8],
|
||||
i1: usize,
|
||||
i2: usize,
|
||||
) -> bool {
|
||||
use core::arch::x86_64::*;
|
||||
|
||||
let n = needle_lower.len();
|
||||
let hlen = haystack.len();
|
||||
let ptr = haystack.as_ptr();
|
||||
let last_start = hlen - n; // last valid match-start position
|
||||
|
||||
let b1 = needle_lower[i1];
|
||||
let b1_alt = if b1.is_ascii_lowercase() {
|
||||
ascii_swap_case(b1)
|
||||
} else {
|
||||
b1
|
||||
};
|
||||
let b2 = needle_lower[i2];
|
||||
let b2_alt = if b2.is_ascii_lowercase() {
|
||||
ascii_swap_case(b2)
|
||||
} else {
|
||||
b2
|
||||
};
|
||||
|
||||
let v1_lo = _mm256_set1_epi8(b1 as i8);
|
||||
let v1_hi = _mm256_set1_epi8(b1_alt as i8);
|
||||
let v2_lo = _mm256_set1_epi8(b2 as i8);
|
||||
let v2_hi = _mm256_set1_epi8(b2_alt as i8);
|
||||
|
||||
// Main loop: process 32 candidate positions per iteration.
|
||||
// We load from ptr+offset+i1 and ptr+offset+i2, so we need
|
||||
// offset + max(i1,i2) + 31 < hlen.
|
||||
let max_idx = i1.max(i2);
|
||||
let max_offset = hlen.saturating_sub(max_idx + 32);
|
||||
let mut offset = 0usize;
|
||||
|
||||
while offset <= max_offset {
|
||||
let chunk1 = unsafe { _mm256_loadu_si256(ptr.add(offset + i1) as *const __m256i) };
|
||||
let chunk2 = unsafe { _mm256_loadu_si256(ptr.add(offset + i2) as *const __m256i) };
|
||||
|
||||
// Case-insensitive match: OR both case variants, then AND the two positions.
|
||||
let eq1 = _mm256_or_si256(
|
||||
_mm256_cmpeq_epi8(chunk1, v1_lo),
|
||||
_mm256_cmpeq_epi8(chunk1, v1_hi),
|
||||
);
|
||||
let eq2 = _mm256_or_si256(
|
||||
_mm256_cmpeq_epi8(chunk2, v2_lo),
|
||||
_mm256_cmpeq_epi8(chunk2, v2_hi),
|
||||
);
|
||||
|
||||
let mut mask = _mm256_movemask_epi8(_mm256_and_si256(eq1, eq2)) as u32;
|
||||
|
||||
while mask != 0 {
|
||||
let bit = mask.trailing_zeros() as usize;
|
||||
let candidate = offset + bit;
|
||||
if candidate > last_start {
|
||||
// Past the end — no more valid positions in this or future chunks.
|
||||
return false;
|
||||
}
|
||||
if unsafe { verify_dispatch(ptr.add(candidate), needle_lower) } {
|
||||
return true;
|
||||
}
|
||||
mask &= mask - 1;
|
||||
}
|
||||
|
||||
offset += 32;
|
||||
}
|
||||
|
||||
// Tail: remaining positions that couldn't fill a full vector.
|
||||
// Use memchr2 on the rarest byte for these last few positions.
|
||||
if offset <= last_start {
|
||||
let rare_pos =
|
||||
if case_insensitive_rank(needle_lower[i1]) <= case_insensitive_rank(needle_lower[i2]) {
|
||||
i1
|
||||
} else {
|
||||
i2
|
||||
};
|
||||
let rare_byte = needle_lower[rare_pos];
|
||||
let tail_start = offset + rare_pos;
|
||||
let tail_end = last_start + rare_pos + 1;
|
||||
if tail_start < tail_end {
|
||||
let tail_space = &haystack[tail_start..tail_end];
|
||||
if rare_byte.is_ascii_lowercase() {
|
||||
for pos in memchr::memchr2_iter(rare_byte, ascii_swap_case(rare_byte), tail_space) {
|
||||
let candidate = offset + pos;
|
||||
if unsafe { verify_dispatch(ptr.add(candidate), needle_lower) } {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
} else {
|
||||
for pos in memchr::memchr_iter(rare_byte, tail_space) {
|
||||
let candidate = offset + pos;
|
||||
if unsafe { verify_dispatch(ptr.add(candidate), needle_lower) } {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
false
|
||||
}
|
||||
|
||||
/// Packed-pair case-insensitive substring search.
|
||||
///
|
||||
/// Selects the two rarest bytes from the needle (using the memchr byte
|
||||
/// frequency heuristic), then SIMD-scans for both at their known offsets
|
||||
/// simultaneously. Falls back to `search` for needles shorter than 2 bytes.
|
||||
pub fn search_packed_pair(haystack: &[u8], needle_lower: &[u8]) -> bool {
|
||||
let n = needle_lower.len();
|
||||
if n == 0 {
|
||||
return true;
|
||||
}
|
||||
if n < 2 {
|
||||
return search(haystack, needle_lower);
|
||||
}
|
||||
if n > haystack.len() {
|
||||
return false;
|
||||
}
|
||||
|
||||
let (i1, i2) = select_rare_pair(needle_lower);
|
||||
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
{
|
||||
if std::is_x86_feature_detected!("avx2") {
|
||||
// Need enough haystack for at least one vector load.
|
||||
let max_idx = i1.max(i2);
|
||||
if haystack.len() >= max_idx + 32 {
|
||||
return unsafe { search_packed_pair_avx2(haystack, needle_lower, i1, i2) };
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
{
|
||||
// The NEON packed-pair scan checks 16 bytes/iteration with ~7 ops,
|
||||
// while memchr's optimized loop processes more bytes with fewer ops.
|
||||
// Packed-pair wins when the first byte is common (lots of false
|
||||
// positives for memchr2 that we avoid). But when the first byte is
|
||||
// rare (z, q, x, ...) memchr2 has no false positives and its raw
|
||||
// throughput dominates. Threshold 200 on the frequency table splits
|
||||
// common letters (s=243, e=253, f=227) from rare ones (z=152, q=139).
|
||||
let first_byte_rank = case_insensitive_rank(needle_lower[0]);
|
||||
let max_idx = i1.max(i2);
|
||||
if first_byte_rank >= 200 && haystack.len() >= max_idx + 16 {
|
||||
return unsafe { search_packed_pair_neon(haystack, needle_lower, i1, i2) };
|
||||
}
|
||||
}
|
||||
|
||||
// Fallback for short haystacks or non-SIMD platforms.
|
||||
search(haystack, needle_lower)
|
||||
}
|
||||
|
||||
// ── Original memchr2 first-byte search ──────────────────────────────────
|
||||
|
||||
/// Case-insensitive search using memchr2 on the first byte.
|
||||
pub fn search(haystack: &[u8], needle_lower: &[u8]) -> bool {
|
||||
let n = needle_lower.len();
|
||||
if n == 0 {
|
||||
return true;
|
||||
}
|
||||
if n > haystack.len() {
|
||||
return false;
|
||||
}
|
||||
|
||||
let search_space = &haystack[..=haystack.len() - n];
|
||||
let first = needle_lower[0];
|
||||
|
||||
if first.is_ascii_lowercase() {
|
||||
let alt = ascii_swap_case(first);
|
||||
for pos in memchr::memchr2_iter(first, alt, search_space) {
|
||||
if unsafe { verify_dispatch(haystack.as_ptr().add(pos), needle_lower) } {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
} else {
|
||||
for pos in memchr::memchr_iter(first, search_space) {
|
||||
if unsafe { verify_dispatch(haystack.as_ptr().add(pos), needle_lower) } {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
false
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn basic_case_insensitive() {
|
||||
assert!(search_packed_pair(b"Hello World", b"hello"));
|
||||
assert!(search_packed_pair(b"Hello World", b"world"));
|
||||
assert!(search_packed_pair(b"NOMORE bugs", b"nomore"));
|
||||
assert!(!search_packed_pair(b"Hello World", b"xyz"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn edge_cases() {
|
||||
assert!(search_packed_pair(b"ab", b"ab"));
|
||||
assert!(search_packed_pair(b"AB", b"ab"));
|
||||
assert!(!search_packed_pair(b"a", b"ab"));
|
||||
assert!(search_packed_pair(b"anything", b""));
|
||||
assert!(!search_packed_pair(b"", b"x"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn packed_pair_matches_search() {
|
||||
let haystacks: &[&[u8]] = &[
|
||||
b"The quick brown fox jumps over the lazy dog",
|
||||
b"int mutex_lock(struct mutex *lock) { return 0; }",
|
||||
b"#define NOMORE_RETRIES 5\nif (nomore) return;",
|
||||
b"abcdefghijklmnopqrstuvwxyz",
|
||||
b"short",
|
||||
];
|
||||
let needles: &[&[u8]] = &[b"fox", b"mutex", b"nomore", b"xyz", b"the", b"short", b"qr"];
|
||||
for h in haystacks {
|
||||
for n in needles {
|
||||
let lower: Vec<u8> = n.iter().map(|b| b.to_ascii_lowercase()).collect();
|
||||
assert_eq!(
|
||||
search_packed_pair(h, &lower),
|
||||
search(h, &lower),
|
||||
"mismatch for haystack={:?} needle={:?}",
|
||||
std::str::from_utf8(h),
|
||||
std::str::from_utf8(n),
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn long_haystack_neon_path() {
|
||||
// Haystack > 16 bytes exercises NEON packed-pair search loop
|
||||
let haystack =
|
||||
b"aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaTHIS_IS_A_LONG_NEEDLE_TESTbbbbbbbbbbbbbbbbbb";
|
||||
assert!(search_packed_pair(haystack, b"this_is_a_long_needle_test"));
|
||||
assert!(!search_packed_pair(
|
||||
haystack,
|
||||
b"this_is_a_long_needle_testz"
|
||||
));
|
||||
|
||||
// Needle >= 16 bytes exercises NEON dotprod verify
|
||||
let long_needle = b"struct mutex *lock";
|
||||
let haystack2 = b"int STRUCT MUTEX *LOCK(struct mutex *lock) { return 0; }";
|
||||
assert!(search_packed_pair(haystack2, long_needle));
|
||||
|
||||
// All uppercase haystack, lowercase needle
|
||||
let upper_hay = b"ABCDEFGHIJKLMNOPQRSTUVWXYZ0123456789ABCDEFGHIJKLMNOPQRSTUVWXYZ";
|
||||
assert!(search_packed_pair(upper_hay, b"qrstuvwxyz0123456789a"));
|
||||
assert!(!search_packed_pair(upper_hay, b"qrstuvwxyz01234567899"));
|
||||
|
||||
// Needle at very end
|
||||
let end_hay = b"xxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxfind_me";
|
||||
assert!(search_packed_pair(end_hay, b"find_me"));
|
||||
|
||||
// Needle at very start
|
||||
assert!(search_packed_pair(end_hay, b"xx"));
|
||||
|
||||
// 1KB haystack with needle near the end
|
||||
let mut big = vec![b'z'; 1024];
|
||||
big[1000..1010].copy_from_slice(b"hElLo_WoRl");
|
||||
assert!(search_packed_pair(&big, b"hello_wo"));
|
||||
assert!(!search_packed_pair(&big, b"hello_world"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn rare_pair_selection() {
|
||||
// For "nomore": n=246, o=244, m=233, o=244, r=245, e=253
|
||||
// Rarest positions should include 'm' (pos 2, rank 233)
|
||||
let (i1, i2) = select_rare_pair(b"nomore");
|
||||
let ranks: Vec<u8> = b"nomore"
|
||||
.iter()
|
||||
.map(|&b| case_insensitive_rank(b))
|
||||
.collect();
|
||||
let r1 = ranks[i1];
|
||||
let r2 = ranks[i2];
|
||||
// Both selected ranks should be <= all other ranks
|
||||
for (i, &r) in ranks.iter().enumerate() {
|
||||
if i != i1 && i != i2 {
|
||||
assert!(r1 <= r || r2 <= r, "pair ({i1},{i2}) not optimal");
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
+573
-112
@@ -1,75 +1,149 @@
|
||||
//! Constraint filtering engine for fff.
|
||||
//!
|
||||
//! This module provides the core constraint application logic that filters items
|
||||
//! based on parsed query constraints (extensions, path segments, globs, git status, etc.).
|
||||
//!
|
||||
//! The filtering is generic over the [`Constrainable`] trait, allowing reuse across
|
||||
//! different search modes (file picker, live grep, etc.).
|
||||
//! Constraint-based prefiltering for search queries.
|
||||
|
||||
use ahash::AHashSet;
|
||||
use fff_query_parser::{Constraint, GitStatusFilter};
|
||||
use smallvec::SmallVec;
|
||||
use zlob::{ZlobFlags, zlob_match_paths};
|
||||
|
||||
use crate::git::is_modified_status;
|
||||
use crate::simd_path::ArenaPtr;
|
||||
|
||||
/// Minimum item count before switching to parallel iteration with rayon.
|
||||
/// Below this threshold, the overhead of thread pool dispatch outweighs the benefit.
|
||||
const PAR_THRESHOLD: usize = 10_000;
|
||||
|
||||
/// Trait for items that can be filtered by constraints.
|
||||
/// Implement this for any searchable item type (files, grep results, etc.).
|
||||
pub trait Constrainable {
|
||||
/// The file's relative path (e.g. "src/main.rs")
|
||||
fn relative_path(&self) -> &str;
|
||||
|
||||
/// The file's lowercased relative path for case-insensitive matching
|
||||
fn relative_path_lower(&self) -> &str;
|
||||
|
||||
/// The file name component (e.g. "main.rs")
|
||||
fn file_name(&self) -> &str;
|
||||
|
||||
/// The git status of this item, if available
|
||||
fn git_status(&self) -> Option<git2::Status>;
|
||||
}
|
||||
|
||||
/// Check if file extension matches (without allocation)
|
||||
/// `needle` must already be lowercase.
|
||||
#[inline]
|
||||
pub fn file_has_extension(file_name: &str, ext: &str) -> bool {
|
||||
if file_name.len() <= ext.len() + 1 {
|
||||
fn contains_ascii_ci(haystack: &str, needle: &str) -> bool {
|
||||
let h = haystack.as_bytes();
|
||||
let n = needle.as_bytes();
|
||||
if n.len() > h.len() {
|
||||
return false;
|
||||
}
|
||||
let start = file_name.len() - ext.len() - 1;
|
||||
file_name.as_bytes().get(start) == Some(&b'.')
|
||||
&& file_name[start + 1..].eq_ignore_ascii_case(ext)
|
||||
if n.is_empty() {
|
||||
return true;
|
||||
}
|
||||
let first = n[0];
|
||||
for i in 0..=(h.len() - n.len()) {
|
||||
if h[i].to_ascii_lowercase() == first
|
||||
&& h[i..i + n.len()]
|
||||
.iter()
|
||||
.zip(n)
|
||||
.all(|(a, b)| a.to_ascii_lowercase() == *b)
|
||||
{
|
||||
return true;
|
||||
}
|
||||
}
|
||||
false
|
||||
}
|
||||
|
||||
/// Check if path contains segment (without allocation)
|
||||
const PAR_THRESHOLD: usize = 10_000;
|
||||
|
||||
pub(crate) trait Constrainable {
|
||||
fn write_file_name(&self, arena: ArenaPtr, out: &mut String);
|
||||
fn git_status(&self) -> Option<git2::Status>;
|
||||
fn write_relative_path(&self, arena: ArenaPtr, out: &mut String);
|
||||
}
|
||||
|
||||
/// Windows stores paths with `\\`; `/` comes from user queries.
|
||||
#[inline]
|
||||
fn is_path_sep(b: u8) -> bool {
|
||||
#[cfg(windows)]
|
||||
{
|
||||
b == b'/' || b == b'\\'
|
||||
}
|
||||
#[cfg(not(windows))]
|
||||
{
|
||||
b == b'/'
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
fn path_slice_eq(a: &[u8], b: &[u8]) -> bool {
|
||||
if a.len() != b.len() {
|
||||
return false;
|
||||
}
|
||||
a.iter().zip(b).all(|(x, y)| {
|
||||
if is_path_sep(*x) && is_path_sep(*y) {
|
||||
true
|
||||
} else {
|
||||
x.eq_ignore_ascii_case(y)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
/// Path ends with suffix at a path-separator boundary (case-insensitive).
|
||||
#[inline]
|
||||
pub fn path_ends_with_suffix(path: &str, suffix: &str) -> bool {
|
||||
let path_bytes = path.as_bytes();
|
||||
let suffix_bytes = suffix.as_bytes();
|
||||
if path_bytes.len() < suffix_bytes.len() {
|
||||
return false;
|
||||
}
|
||||
|
||||
let start = path.len() - suffix.len();
|
||||
|
||||
// Multi-byte UTF-8 may put `start` inside a char.
|
||||
if !path.is_char_boundary(start) {
|
||||
return false;
|
||||
}
|
||||
|
||||
if !path_slice_eq(&path_bytes[start..], suffix_bytes) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Exact or preceded by a separator. Scan backward past any multi-byte
|
||||
// continuation bytes to find the preceding ASCII byte.
|
||||
if start == 0 {
|
||||
return true;
|
||||
}
|
||||
let mut i = start;
|
||||
while i > 0 {
|
||||
i -= 1;
|
||||
if path_bytes[i] < 128 {
|
||||
return is_path_sep(path_bytes[i]);
|
||||
}
|
||||
}
|
||||
false
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub fn file_has_extension(file_name: &str, ext: &str) -> bool {
|
||||
let name_bytes = file_name.as_bytes();
|
||||
let ext_bytes = ext.as_bytes();
|
||||
if name_bytes.len() <= ext_bytes.len() + 1 {
|
||||
return false;
|
||||
}
|
||||
let start = name_bytes.len() - ext_bytes.len() - 1;
|
||||
if start > 0 && !file_name.is_char_boundary(start) {
|
||||
return false;
|
||||
}
|
||||
name_bytes.get(start) == Some(&b'.') && name_bytes[start + 1..].eq_ignore_ascii_case(ext_bytes)
|
||||
}
|
||||
|
||||
/// Matches multi-segment queries like `libswscale/aarch64`.
|
||||
#[inline]
|
||||
pub fn path_contains_segment(path: &str, segment: &str) -> bool {
|
||||
let path_bytes = path.as_bytes();
|
||||
let segment_len = segment.len();
|
||||
let segment_bytes = segment.as_bytes();
|
||||
let segment_len = segment_bytes.len();
|
||||
|
||||
// Check segment/ at start
|
||||
if path.len() > segment_len
|
||||
&& path_bytes.get(segment_len) == Some(&b'/')
|
||||
&& path[..segment_len].eq_ignore_ascii_case(segment)
|
||||
if path_bytes.len() > segment_len
|
||||
&& is_path_sep(path_bytes[segment_len])
|
||||
&& path.is_char_boundary(segment_len)
|
||||
&& path_slice_eq(&path_bytes[..segment_len], segment_bytes)
|
||||
{
|
||||
return true;
|
||||
}
|
||||
|
||||
// Check /segment/ anywhere using byte scanning
|
||||
if path.len() < segment_len + 2 {
|
||||
if path_bytes.len() < segment_len + 2 {
|
||||
return false;
|
||||
}
|
||||
|
||||
for i in 0..path.len().saturating_sub(segment_len + 1) {
|
||||
if path_bytes[i] == b'/' {
|
||||
for i in 0..path_bytes.len().saturating_sub(segment_len + 1) {
|
||||
if is_path_sep(path_bytes[i]) {
|
||||
let start = i + 1;
|
||||
let end = start + segment_len;
|
||||
if end < path.len()
|
||||
&& path_bytes[end] == b'/'
|
||||
&& path[start..end].eq_ignore_ascii_case(segment)
|
||||
if end < path_bytes.len()
|
||||
&& is_path_sep(path_bytes[end])
|
||||
&& path.is_char_boundary(start)
|
||||
&& path.is_char_boundary(end)
|
||||
&& path_slice_eq(&path_bytes[start..end], segment_bytes)
|
||||
{
|
||||
return true;
|
||||
}
|
||||
@@ -78,8 +152,8 @@ pub fn path_contains_segment(path: &str, segment: &str) -> bool {
|
||||
false
|
||||
}
|
||||
|
||||
/// Check if an item at given index matches a constraint (single-pass friendly, allocation-free)
|
||||
#[inline]
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
fn item_matches_constraint_at_index<T: Constrainable>(
|
||||
item: &T,
|
||||
item_index: usize,
|
||||
@@ -87,21 +161,35 @@ fn item_matches_constraint_at_index<T: Constrainable>(
|
||||
glob_results: &[(bool, AHashSet<usize>)],
|
||||
glob_idx: &mut usize,
|
||||
negate: bool,
|
||||
arena: ArenaPtr,
|
||||
fname_buf: &mut String,
|
||||
path_buf: &mut String,
|
||||
) -> bool {
|
||||
let matches = match constraint {
|
||||
Constraint::Extension(ext) => file_has_extension(item.file_name(), ext),
|
||||
Constraint::Extension(ext) => {
|
||||
item.write_file_name(arena, fname_buf);
|
||||
file_has_extension(fname_buf, ext)
|
||||
}
|
||||
Constraint::Glob(_) => {
|
||||
let result = glob_results
|
||||
.get(*glob_idx)
|
||||
.map(|(is_neg, set)| {
|
||||
let matched = set.contains(&item_index);
|
||||
|
||||
if *is_neg { !matched } else { matched }
|
||||
})
|
||||
.unwrap_or(true);
|
||||
*glob_idx += 1;
|
||||
return if negate { !result } else { result };
|
||||
}
|
||||
Constraint::PathSegment(segment) => path_contains_segment(item.relative_path(), segment),
|
||||
Constraint::PathSegment(segment) => {
|
||||
item.write_relative_path(arena, path_buf);
|
||||
path_contains_segment(path_buf, segment)
|
||||
}
|
||||
Constraint::FilePath(suffix) => {
|
||||
item.write_relative_path(arena, path_buf);
|
||||
path_ends_with_suffix(path_buf, suffix)
|
||||
}
|
||||
Constraint::GitStatus(status_filter) => match (item.git_status(), status_filter) {
|
||||
(Some(status), GitStatusFilter::Modified) => is_modified_status(status),
|
||||
(Some(status), GitStatusFilter::Untracked) => status.contains(git2::Status::WT_NEW),
|
||||
@@ -124,11 +212,17 @@ fn item_matches_constraint_at_index<T: Constrainable>(
|
||||
glob_results,
|
||||
glob_idx,
|
||||
!negate,
|
||||
arena,
|
||||
fname_buf,
|
||||
path_buf,
|
||||
);
|
||||
}
|
||||
|
||||
// only works with negation
|
||||
Constraint::Text(text) => item.relative_path_lower().contains(text),
|
||||
Constraint::Text(text) => {
|
||||
item.write_relative_path(arena, path_buf);
|
||||
contains_ascii_ci(path_buf, text)
|
||||
}
|
||||
|
||||
// Parts and Exclude are handled at a higher level
|
||||
Constraint::Parts(_) | Constraint::Exclude(_) | Constraint::FileType(_) => true,
|
||||
@@ -137,15 +231,12 @@ fn item_matches_constraint_at_index<T: Constrainable>(
|
||||
if negate { !matches } else { matches }
|
||||
}
|
||||
|
||||
/// Apply constraint-based prefiltering in a single pass over all items.
|
||||
/// Returns `None` if no constraints are present, `Some(filtered)` otherwise.
|
||||
/// Multiple extension constraints (*.rs *.ts) are combined with OR logic.
|
||||
/// All other constraints are combined with AND logic.
|
||||
///
|
||||
/// Uses parallel iteration via rayon when the item count exceeds [`PAR_THRESHOLD`].
|
||||
pub fn apply_constraints<'a, T: Constrainable + Sync>(
|
||||
/// Extension constraints use OR logic; all others use AND.
|
||||
pub(crate) fn apply_constraints<'a, T: Constrainable + Sync>(
|
||||
items: &'a [T],
|
||||
constraints: &[Constraint<'_>],
|
||||
arena: ArenaPtr,
|
||||
) -> Option<Vec<&'a T>> {
|
||||
if constraints.is_empty() {
|
||||
return None;
|
||||
@@ -168,47 +259,107 @@ pub fn apply_constraints<'a, T: Constrainable + Sync>(
|
||||
.any(|c| matches!(c, Constraint::Glob(_) | Constraint::Not(_)));
|
||||
|
||||
let glob_results = if has_globs {
|
||||
let paths: Vec<&str> = items.iter().map(|f| f.relative_path()).collect();
|
||||
precompute_glob_matches(&other_constraints, &paths)
|
||||
// Build a single contiguous buffer of all relative paths + offset table.
|
||||
// One allocation for the buffer, one for offsets — NOT one String per file.
|
||||
// On Windows we fold `\\` into `/` while copying so globset/zlob see a
|
||||
// canonical separator. The rewrite is in place on bytes we just wrote.
|
||||
let mut path_buf = Vec::<u8>::new();
|
||||
let mut offsets = Vec::<(usize, usize)>::with_capacity(items.len());
|
||||
let mut tmp = String::with_capacity(64);
|
||||
for item in items.iter() {
|
||||
let start = path_buf.len();
|
||||
item.write_relative_path(arena, &mut tmp);
|
||||
path_buf.extend_from_slice(tmp.as_bytes());
|
||||
#[cfg(windows)]
|
||||
for b in &mut path_buf[start..] {
|
||||
if *b == b'\\' {
|
||||
*b = b'/';
|
||||
}
|
||||
}
|
||||
offsets.push((start, path_buf.len() - start));
|
||||
}
|
||||
let path_refs: Vec<&str> = offsets
|
||||
.iter()
|
||||
.map(|&(off, len)| unsafe { std::str::from_utf8_unchecked(&path_buf[off..off + len]) })
|
||||
.collect();
|
||||
precompute_glob_matches(&other_constraints, &path_refs)
|
||||
} else {
|
||||
Vec::new()
|
||||
};
|
||||
|
||||
let matches_constraints = |i: usize, item: &T| -> bool {
|
||||
if !extensions.is_empty()
|
||||
&& !extensions
|
||||
.iter()
|
||||
.any(|ext| file_has_extension(item.file_name(), ext))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
let mut glob_idx = 0;
|
||||
other_constraints.iter().all(|constraint| {
|
||||
item_matches_constraint_at_index(
|
||||
item,
|
||||
i,
|
||||
constraint,
|
||||
&glob_results,
|
||||
&mut glob_idx,
|
||||
false,
|
||||
)
|
||||
})
|
||||
};
|
||||
|
||||
let filtered: Vec<&T> = if items.len() >= PAR_THRESHOLD {
|
||||
use rayon::prelude::*;
|
||||
items
|
||||
.par_iter()
|
||||
.enumerate()
|
||||
.filter(|(i, item)| matches_constraints(*i, item))
|
||||
.map(|(_, item)| item)
|
||||
.map_init(
|
||||
|| (String::with_capacity(64), String::with_capacity(64)),
|
||||
|(fname_buf, path_buf), (i, item)| {
|
||||
if !extensions.is_empty() {
|
||||
item.write_file_name(arena, fname_buf);
|
||||
if !extensions
|
||||
.iter()
|
||||
.any(|ext| file_has_extension(fname_buf, ext))
|
||||
{
|
||||
return None;
|
||||
}
|
||||
}
|
||||
|
||||
let mut glob_idx = 0;
|
||||
if other_constraints.iter().all(|constraint| {
|
||||
item_matches_constraint_at_index(
|
||||
item,
|
||||
i,
|
||||
constraint,
|
||||
&glob_results,
|
||||
&mut glob_idx,
|
||||
false,
|
||||
arena,
|
||||
fname_buf,
|
||||
path_buf,
|
||||
)
|
||||
}) {
|
||||
Some(item)
|
||||
} else {
|
||||
None
|
||||
}
|
||||
},
|
||||
)
|
||||
.flatten()
|
||||
.collect()
|
||||
} else {
|
||||
let mut fname_buf = String::with_capacity(64);
|
||||
let mut path_buf = String::with_capacity(64);
|
||||
|
||||
items
|
||||
.iter()
|
||||
.enumerate()
|
||||
.filter(|(i, item)| matches_constraints(*i, item))
|
||||
.filter(|&(i, item)| {
|
||||
if !extensions.is_empty() {
|
||||
item.write_file_name(arena, &mut fname_buf);
|
||||
if !extensions
|
||||
.iter()
|
||||
.any(|ext| file_has_extension(&fname_buf, ext))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
let mut glob_idx = 0;
|
||||
other_constraints.iter().all(|constraint| {
|
||||
item_matches_constraint_at_index(
|
||||
item,
|
||||
i,
|
||||
constraint,
|
||||
&glob_results,
|
||||
&mut glob_idx,
|
||||
false,
|
||||
arena,
|
||||
&mut fname_buf,
|
||||
&mut path_buf,
|
||||
)
|
||||
})
|
||||
})
|
||||
.map(|(_, item)| item)
|
||||
.collect()
|
||||
};
|
||||
@@ -231,48 +382,110 @@ fn collect_glob_indices<'a>(
|
||||
constraint: &Constraint<'a>,
|
||||
paths: &[&str],
|
||||
results: &mut Vec<(bool, AHashSet<usize>)>,
|
||||
is_negated: bool,
|
||||
_is_negated: bool,
|
||||
) {
|
||||
match constraint {
|
||||
Constraint::Glob(pattern) => {
|
||||
if let Ok(Some(matches)) = zlob_match_paths(pattern, paths, ZlobFlags::RECOMMENDED) {
|
||||
let matched_set: AHashSet<usize> =
|
||||
matches.iter().map(|s| s.as_ptr() as usize).collect();
|
||||
|
||||
let indices: AHashSet<usize> = if paths.len() >= PAR_THRESHOLD {
|
||||
use rayon::prelude::*;
|
||||
paths
|
||||
.par_iter()
|
||||
.enumerate()
|
||||
.filter(|(_, p)| matched_set.contains(&(p.as_ptr() as usize)))
|
||||
.map(|(i, _)| i)
|
||||
.collect::<Vec<_>>()
|
||||
.into_iter()
|
||||
.collect()
|
||||
} else {
|
||||
paths
|
||||
.iter()
|
||||
.enumerate()
|
||||
.filter(|(_, p)| matched_set.contains(&(p.as_ptr() as usize)))
|
||||
.map(|(i, _)| i)
|
||||
.collect()
|
||||
};
|
||||
results.push((is_negated, indices));
|
||||
} else {
|
||||
results.push((is_negated, AHashSet::new()));
|
||||
}
|
||||
let indices = match_glob_pattern(pattern, paths);
|
||||
// Negation is handled by the `negate` parameter in
|
||||
// `item_matches_constraint_at_index`, NOT here. Storing
|
||||
// `is_negated=true` caused a double-negation bug when the
|
||||
// Glob arm also applied `negate`.
|
||||
results.push((false, indices));
|
||||
}
|
||||
Constraint::Not(inner) => {
|
||||
collect_glob_indices(inner, paths, results, !is_negated);
|
||||
collect_glob_indices(inner, paths, results, true);
|
||||
}
|
||||
_ => {}
|
||||
}
|
||||
}
|
||||
|
||||
/// Match a glob pattern against a list of paths, returning the set of matching indices.
|
||||
///
|
||||
/// When the `zlob` feature is enabled, delegates to `zlob::zlob_match_paths` (Zig-compiled
|
||||
/// C library, fastest). Otherwise falls back to `globset::Glob` (pure Rust).
|
||||
#[cfg(feature = "zlob")]
|
||||
fn match_glob_pattern(pattern: &str, paths: &[&str]) -> AHashSet<usize> {
|
||||
let Ok(Some(matches)) = zlob::zlob_match_paths(pattern, paths, zlob::ZlobFlags::RECOMMENDED)
|
||||
else {
|
||||
return AHashSet::new();
|
||||
};
|
||||
|
||||
let matched_set: AHashSet<usize> = matches.iter().map(|s| s.as_ptr() as usize).collect();
|
||||
|
||||
if paths.len() >= PAR_THRESHOLD {
|
||||
use rayon::prelude::*;
|
||||
paths
|
||||
.par_iter()
|
||||
.enumerate()
|
||||
.filter(|(_, p)| matched_set.contains(&(p.as_ptr() as usize)))
|
||||
.map(|(i, _)| i)
|
||||
.collect::<Vec<_>>()
|
||||
.into_iter()
|
||||
.collect()
|
||||
} else {
|
||||
paths
|
||||
.iter()
|
||||
.enumerate()
|
||||
.filter(|(_, p)| matched_set.contains(&(p.as_ptr() as usize)))
|
||||
.map(|(i, _)| i)
|
||||
.collect()
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(not(feature = "zlob"))]
|
||||
fn match_glob_pattern(pattern: &str, paths: &[&str]) -> AHashSet<usize> {
|
||||
let Ok(glob) = globset::Glob::new(pattern) else {
|
||||
return AHashSet::new();
|
||||
};
|
||||
let matcher = glob.compile_matcher();
|
||||
|
||||
if paths.len() >= PAR_THRESHOLD {
|
||||
use rayon::prelude::*;
|
||||
paths
|
||||
.par_iter()
|
||||
.enumerate()
|
||||
.filter(|(_, p)| matcher.is_match(p))
|
||||
.map(|(i, _)| i)
|
||||
.collect::<Vec<_>>()
|
||||
.into_iter()
|
||||
.collect()
|
||||
} else {
|
||||
paths
|
||||
.iter()
|
||||
.enumerate()
|
||||
.filter(|(_, p)| matcher.is_match(p))
|
||||
.map(|(i, _)| i)
|
||||
.collect()
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[derive(Clone)]
|
||||
struct TestItem {
|
||||
relative_path: &'static str,
|
||||
file_name: &'static str,
|
||||
}
|
||||
|
||||
impl Constrainable for TestItem {
|
||||
fn write_file_name(&self, _arena: ArenaPtr, out: &mut String) {
|
||||
out.clear();
|
||||
out.push_str(self.file_name);
|
||||
}
|
||||
|
||||
fn write_relative_path(&self, _arena: ArenaPtr, out: &mut String) {
|
||||
out.clear();
|
||||
out.push_str(self.relative_path);
|
||||
}
|
||||
|
||||
fn git_status(&self) -> Option<git2::Status> {
|
||||
None
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_file_has_extension() {
|
||||
assert!(file_has_extension("file.rs", "rs"));
|
||||
@@ -312,8 +525,256 @@ mod tests {
|
||||
// Should not match filename
|
||||
assert!(!path_contains_segment("lib/src", "src"));
|
||||
|
||||
// Multi-segment constraints
|
||||
assert!(path_contains_segment(
|
||||
"libswscale/aarch64/input.S",
|
||||
"libswscale/aarch64"
|
||||
));
|
||||
assert!(path_contains_segment(
|
||||
"foo/libswscale/aarch64/input.S",
|
||||
"libswscale/aarch64"
|
||||
));
|
||||
assert!(path_contains_segment(
|
||||
"foo/LibSwscale/AArch64/input.S",
|
||||
"libswscale/aarch64"
|
||||
)); // case-insensitive
|
||||
assert!(!path_contains_segment(
|
||||
"xlibswscale/aarch64/input.S",
|
||||
"libswscale/aarch64"
|
||||
)); // partial match at start
|
||||
assert!(!path_contains_segment(
|
||||
"foo/libswscale/aarch64x/input.S",
|
||||
"libswscale/aarch64"
|
||||
)); // partial match at end
|
||||
assert!(path_contains_segment(
|
||||
"crates/fff-core/src/grep.rs",
|
||||
"fff-core/src"
|
||||
));
|
||||
|
||||
// Edge cases
|
||||
assert!(!path_contains_segment("", "src"));
|
||||
assert!(!path_contains_segment("src", "src")); // no trailing slash
|
||||
}
|
||||
|
||||
#[cfg(windows)]
|
||||
#[test]
|
||||
fn test_path_contains_segment_accepts_backslash() {
|
||||
assert!(path_contains_segment("src\\lib.rs", "src"));
|
||||
assert!(path_contains_segment(
|
||||
"app\\modules\\src\\services\\x.lua",
|
||||
"src"
|
||||
));
|
||||
assert!(path_contains_segment("app\\SRC\\x.lua", "src"));
|
||||
|
||||
assert!(path_contains_segment(
|
||||
"foo\\libswscale\\aarch64\\input.S",
|
||||
"libswscale/aarch64"
|
||||
));
|
||||
assert!(path_contains_segment(
|
||||
"crates\\fff-core\\src\\grep.rs",
|
||||
"fff-core/src"
|
||||
));
|
||||
|
||||
assert!(!path_contains_segment("mysrc\\lib.rs", "src"));
|
||||
assert!(!path_contains_segment(
|
||||
"xlibswscale\\aarch64\\in.S",
|
||||
"libswscale/aarch64"
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_path_ends_with_suffix() {
|
||||
// Exact match
|
||||
assert!(path_ends_with_suffix(
|
||||
"libswscale/input.c",
|
||||
"libswscale/input.c"
|
||||
));
|
||||
|
||||
// Suffix match at / boundary
|
||||
assert!(path_ends_with_suffix(
|
||||
"foo/libswscale/input.c",
|
||||
"libswscale/input.c"
|
||||
));
|
||||
|
||||
// Deep nesting
|
||||
assert!(path_ends_with_suffix(
|
||||
"a/b/c/libswscale/input.c",
|
||||
"libswscale/input.c"
|
||||
));
|
||||
|
||||
// No boundary — partial directory name
|
||||
assert!(!path_ends_with_suffix(
|
||||
"xlibswscale/input.c",
|
||||
"libswscale/input.c"
|
||||
));
|
||||
|
||||
// Case insensitive
|
||||
assert!(path_ends_with_suffix(
|
||||
"foo/LibSwscale/Input.C",
|
||||
"libswscale/input.c"
|
||||
));
|
||||
|
||||
// Single file name
|
||||
assert!(path_ends_with_suffix("input.c", "input.c"));
|
||||
assert!(!path_ends_with_suffix("xinput.c", "input.c"));
|
||||
|
||||
// Suffix longer than path
|
||||
assert!(!path_ends_with_suffix("input.c", "foo/input.c"));
|
||||
|
||||
// Simple path
|
||||
assert!(path_ends_with_suffix("src/main.rs", "src/main.rs"));
|
||||
assert!(path_ends_with_suffix("crates/src/main.rs", "src/main.rs"));
|
||||
}
|
||||
|
||||
#[cfg(windows)]
|
||||
#[test]
|
||||
fn test_path_ends_with_suffix_accepts_backslash() {
|
||||
assert!(path_ends_with_suffix(
|
||||
"app\\modules\\src\\services\\handler.lua",
|
||||
"services/handler.lua"
|
||||
));
|
||||
assert!(path_ends_with_suffix(
|
||||
"foo\\libswscale\\input.c",
|
||||
"libswscale/input.c"
|
||||
));
|
||||
assert!(!path_ends_with_suffix(
|
||||
"xlibswscale\\input.c",
|
||||
"libswscale/input.c"
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_path_ends_with_suffix_does_not_panic_on_unicode_suffix() {
|
||||
assert!(!path_ends_with_suffix("유니코드_파일_테스트.csv", "트.c"));
|
||||
assert!(path_ends_with_suffix(
|
||||
"data/유니코드_파일_테스트.csv",
|
||||
"유니코드_파일_테스트.csv"
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_path_ends_with_suffix_unicode_apostrophe_mismatch() {
|
||||
assert!(!path_ends_with_suffix(
|
||||
"dir/\u{2019}bar/file.txt",
|
||||
"'bar/file.txt"
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_path_ends_with_suffix_unicode_space_mismatch() {
|
||||
assert!(!path_ends_with_suffix(
|
||||
"dir/\u{202f}am/file.txt",
|
||||
" am/file.txt"
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_path_contains_segment_does_not_panic_on_unicode_segment() {
|
||||
assert!(!path_contains_segment("문서/notes.txt", "문x"));
|
||||
assert!(path_contains_segment("프로젝트/문서/notes.txt", "문서"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_path_contains_segment_unicode_no_panic() {
|
||||
assert!(!path_contains_segment(
|
||||
"Library/Cloud/Project\u{2019}s Folder/books.ttl",
|
||||
"Project's Folder"
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_file_has_extension_unicode_no_panic() {
|
||||
assert!(!file_has_extension("cat\u{00e9}.rs", "s"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_file_has_extension_unicode_filename() {
|
||||
assert!(file_has_extension("운영-가이드.md", "md"));
|
||||
assert!(file_has_extension("테스트.csv", "csv"));
|
||||
assert!(!file_has_extension("테스트.csv", "md"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_apply_constraints_file_path_with_unicode_suffix() {
|
||||
let arena_ptr = ArenaPtr(std::ptr::null());
|
||||
|
||||
let item = TestItem {
|
||||
relative_path: "data/유니코드_파일_테스트.csv",
|
||||
file_name: "유니코드_파일_테스트.csv",
|
||||
};
|
||||
|
||||
let exact = [Constraint::FilePath("유니코드_파일_테스트.csv")];
|
||||
let mismatch = [Constraint::FilePath("트.c")];
|
||||
|
||||
let exact_items = [item.clone()];
|
||||
let exact_matches =
|
||||
apply_constraints(&exact_items, &exact, arena_ptr).expect("constraints applied");
|
||||
assert_eq!(exact_matches.len(), 1);
|
||||
|
||||
let mismatch_items = [item];
|
||||
let mismatch_matches =
|
||||
apply_constraints(&mismatch_items, &mismatch, arena_ptr).expect("constraints applied");
|
||||
assert!(mismatch_matches.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_unicode_path_no_panic_real_korean_cases() {
|
||||
// Real Korean paths that caused panics
|
||||
let path1 = "Downloads/(커리큘럼) hermes agent_정승현님 - 1차 커리큘럼 (강사님 작성).csv";
|
||||
let path2 = "hermes-agent-lecture-materials/세부_커리큘럼_최종.csv";
|
||||
let path3 = "projects/fastcampus-hermes-agent-curriculum/chapters/part-02-Hermes-설치-및-기본-사용/section-02-doctor로-설치-상태-검증/research/03-fix가-자동-수정하는-것과-못하는-것.md";
|
||||
|
||||
// These must not panic regardless of segment/suffix used
|
||||
assert!(!path_contains_segment(path1, "작성"));
|
||||
assert!(!path_ends_with_suffix(path1, "작성.csv"));
|
||||
assert!(!path_contains_segment(path2, "최종"));
|
||||
assert!(!path_ends_with_suffix(path2, "최종.csv"));
|
||||
assert!(!path_contains_segment(path3, "수정"));
|
||||
assert!(!path_ends_with_suffix(path3, "것.md"));
|
||||
|
||||
// Positive cases should still work
|
||||
assert!(path_contains_segment(
|
||||
path2,
|
||||
"hermes-agent-lecture-materials"
|
||||
));
|
||||
assert!(path_ends_with_suffix(
|
||||
path1,
|
||||
"(커리큘럼) hermes agent_정승현님 - 1차 커리큘럼 (강사님 작성).csv"
|
||||
));
|
||||
assert!(path_ends_with_suffix(path2, "세부_커리큘럼_최종.csv"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_negated_glob_excludes_matching_files() {
|
||||
let arena_ptr = ArenaPtr(std::ptr::null());
|
||||
|
||||
let items = vec![
|
||||
TestItem {
|
||||
relative_path: "src/main.rs",
|
||||
file_name: "main.rs",
|
||||
},
|
||||
TestItem {
|
||||
relative_path: "src/lib.ts",
|
||||
file_name: "lib.ts",
|
||||
},
|
||||
TestItem {
|
||||
relative_path: "include/fff.h",
|
||||
file_name: "fff.h",
|
||||
},
|
||||
];
|
||||
|
||||
// Not(Glob("**/*.rs")) should exclude .rs files
|
||||
let constraints = vec![Constraint::Not(Box::new(Constraint::Glob("**/*.rs")))];
|
||||
let result = apply_constraints(&items, &constraints, arena_ptr).unwrap();
|
||||
let paths: Vec<&str> = result.iter().map(|i| i.relative_path).collect();
|
||||
assert!(
|
||||
!paths.contains(&"src/main.rs"),
|
||||
"rs file should be excluded"
|
||||
);
|
||||
assert!(paths.contains(&"src/lib.ts"), "ts file should be included");
|
||||
assert!(
|
||||
paths.contains(&"include/fff.h"),
|
||||
"h file should be included"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -9,16 +9,23 @@ pub struct DbHealth {
|
||||
pub disk_size: u64,
|
||||
/// Entry counts by table name
|
||||
pub entry_counts: Vec<(&'static str, u64)>,
|
||||
/// Set to `false` if can not acquire the write lock
|
||||
pub healthy: bool,
|
||||
}
|
||||
|
||||
pub trait DbHealthChecker {
|
||||
fn get_env(&self) -> &heed::Env;
|
||||
fn is_healthy(&self) -> bool;
|
||||
/// Entries per database, each group has a static string label
|
||||
fn count_entries(&self) -> Result<Vec<(&'static str, u64)>>;
|
||||
|
||||
/// Health summary of the database, returns summary struct
|
||||
fn get_health(&self) -> Result<DbHealth> {
|
||||
let env = self.get_env();
|
||||
|
||||
let size = env.real_disk_size().map_err(crate::error::Error::EnvOpen)?;
|
||||
let size = env
|
||||
.real_disk_size()
|
||||
.map_err(crate::error::Error::GenericDbError)?;
|
||||
let path = env.path().to_string_lossy().to_string();
|
||||
let entry_counts = self.count_entries()?;
|
||||
|
||||
@@ -26,6 +33,7 @@ pub trait DbHealthChecker {
|
||||
path,
|
||||
disk_size: size,
|
||||
entry_counts,
|
||||
healthy: self.is_healthy(),
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,512 @@
|
||||
use super::db_healthcheck::DbHealthChecker;
|
||||
use super::lmdb::{DbHealth, LmdbStore, is_map_full};
|
||||
use crate::error::{Error, Result};
|
||||
use crate::file_picker::FFFMode;
|
||||
use crate::git::is_modified_status;
|
||||
use heed::types::{Bytes, SerdeBincode};
|
||||
use heed::{Database, Env};
|
||||
use std::time::{SystemTime, UNIX_EPOCH};
|
||||
use std::{collections::VecDeque, path::Path};
|
||||
|
||||
const DECAY_CONSTANT: f64 = 0.0693; // ln(2)/10 for 10-day half-life
|
||||
const SECONDS_PER_DAY: f64 = 86400.0;
|
||||
const MAX_HISTORY_DAYS: f64 = 30.0; // Only consider accesses within 30 days
|
||||
const MAX_TIMESTAMPS_PER_FILE: usize = 128;
|
||||
|
||||
// AI mode: faster decay since AI sessions are shorter and more intense
|
||||
const AI_DECAY_CONSTANT: f64 = 0.231; // ln(2)/3 for 3-day half-life
|
||||
const AI_MAX_HISTORY_DAYS: f64 = 7.0; // Only consider accesses within 7 days
|
||||
|
||||
#[derive(Debug)]
|
||||
pub struct FrecencyTracker {
|
||||
env: Env,
|
||||
db: Database<Bytes, SerdeBincode<VecDeque<u64>>>,
|
||||
health: DbHealth,
|
||||
}
|
||||
|
||||
const MODIFICATION_THRESHOLDS: [(i64, u64); 5] = [
|
||||
(16, 60 * 2), // 2 minutes
|
||||
(8, 60 * 15), // 15 minutes
|
||||
(4, 60 * 60), // 1 hour
|
||||
(2, 60 * 60 * 24), // 1 day
|
||||
(1, 60 * 60 * 24 * 7), // 1 week
|
||||
];
|
||||
|
||||
// AI mode: compressed thresholds since AI edits happen in rapid bursts
|
||||
const AI_MODIFICATION_THRESHOLDS: [(i64, u64); 5] = [
|
||||
(16, 30), // 30 seconds
|
||||
(8, 60 * 5), // 5 minutes
|
||||
(4, 60 * 15), // 15 minutes
|
||||
(2, 60 * 60), // 1 hour
|
||||
(1, 60 * 60 * 4), // 4 hours
|
||||
];
|
||||
|
||||
impl DbHealthChecker for FrecencyTracker {
|
||||
fn get_env(&self) -> &heed::Env {
|
||||
&self.env
|
||||
}
|
||||
|
||||
fn is_healthy(&self) -> bool {
|
||||
self.health.is_healthy()
|
||||
}
|
||||
|
||||
fn count_entries(&self) -> Result<Vec<(&'static str, u64)>> {
|
||||
let rtxn = self
|
||||
.env
|
||||
.read_txn()
|
||||
.map_err(|source| Error::DbStartReadTxn {
|
||||
db: Self::LABEL,
|
||||
source,
|
||||
})?;
|
||||
let count = self.db.len(&rtxn).map_err(|source| Error::DbRead {
|
||||
db: Self::LABEL,
|
||||
source,
|
||||
})?;
|
||||
|
||||
Ok(vec![("absolute_frecency_entries", count)])
|
||||
}
|
||||
}
|
||||
|
||||
impl LmdbStore for FrecencyTracker {
|
||||
const LABEL: &'static str = "frecency";
|
||||
// 10 MiB hard ceiling. Owner's db after years of use is ~560 KiB, so this
|
||||
// leaves ~18× headroom while capping runaway growth (see GH issue #437).
|
||||
const MAP_SIZE: usize = 10 * 1024 * 1024;
|
||||
const MAX_DBS: u32 = 0;
|
||||
// Nuke the db when it exceeds 8 MiB on disk — leaves a small margin under
|
||||
// MAP_SIZE so we don't hit MDB_MAP_FULL before the open-time erase fires.
|
||||
const SIZE_CAP_BYTES: u64 = 12 * 1024 * 1024;
|
||||
|
||||
fn env(&self) -> &Env {
|
||||
&self.env
|
||||
}
|
||||
|
||||
fn health(&self) -> &DbHealth {
|
||||
&self.health
|
||||
}
|
||||
|
||||
fn purge_stale_data(env: &Env) -> Result<()> {
|
||||
let (deleted, pruned) = Self::purge_stale_entries(env)?;
|
||||
if deleted > 0 || pruned > 0 {
|
||||
tracing::info!(deleted, pruned, "Frecency GC purged entries");
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
impl FrecencyTracker {
|
||||
/// Returns the on-disk path of the LMDB environment directory.
|
||||
pub fn db_path(&self) -> &Path {
|
||||
self.env.path()
|
||||
}
|
||||
|
||||
pub fn open(db_path: impl AsRef<Path>) -> Result<Self> {
|
||||
let db_path = db_path.as_ref();
|
||||
let (env, health) = Self::open_env(db_path)?;
|
||||
|
||||
let db = Self::open_database_safe(&env, None)?;
|
||||
Ok(FrecencyTracker { db, env, health })
|
||||
}
|
||||
|
||||
#[deprecated(
|
||||
since = "0.7.0",
|
||||
note = "LMDB unsafe no-lock mode is no longer supported; use `FrecencyTracker::open` instead. \
|
||||
The `_use_unsafe_no_lock` argument is ignored."
|
||||
)]
|
||||
pub fn new(db_path: impl AsRef<Path>, _use_unsafe_no_lock: bool) -> Result<Self> {
|
||||
Self::open(db_path)
|
||||
}
|
||||
|
||||
/// Removes entries where all timestamps are older than MAX_HISTORY_DAYS,
|
||||
/// and prunes stale timestamps from entries that still have recent ones.
|
||||
/// Returns (deleted_count, pruned_count).
|
||||
fn purge_stale_entries(env: &Env) -> Result<(usize, usize)> {
|
||||
let now = SystemTime::now()
|
||||
.duration_since(UNIX_EPOCH)
|
||||
.unwrap()
|
||||
.as_secs();
|
||||
let cutoff_time = now.saturating_sub((MAX_HISTORY_DAYS * SECONDS_PER_DAY) as u64);
|
||||
|
||||
let db: Database<Bytes, SerdeBincode<VecDeque<u64>>> = Self::open_database_safe(env, None)?;
|
||||
|
||||
let rtxn = env.read_txn().map_err(|source| Error::DbStartReadTxn {
|
||||
db: Self::LABEL,
|
||||
source,
|
||||
})?;
|
||||
let mut to_delete: Vec<Vec<u8>> = Vec::new();
|
||||
let mut to_update: Vec<(Vec<u8>, VecDeque<u64>)> = Vec::new();
|
||||
|
||||
let iter = db.iter(&rtxn).map_err(|source| Error::DbRead {
|
||||
db: Self::LABEL,
|
||||
source,
|
||||
})?;
|
||||
for result in iter {
|
||||
let (key, accesses) = result.map_err(|source| Error::DbRead {
|
||||
db: Self::LABEL,
|
||||
source,
|
||||
})?;
|
||||
|
||||
// Timestamps chronologically ordered (oldest at front).
|
||||
let fresh_start = accesses.iter().position(|&ts| ts >= cutoff_time);
|
||||
match fresh_start {
|
||||
None => to_delete.push(key.to_vec()),
|
||||
Some(0) => {}
|
||||
Some(start) => {
|
||||
let pruned: VecDeque<u64> = accesses.iter().skip(start).copied().collect();
|
||||
to_update.push((key.to_vec(), pruned));
|
||||
}
|
||||
}
|
||||
}
|
||||
drop(rtxn);
|
||||
|
||||
if to_delete.is_empty() && to_update.is_empty() {
|
||||
return Ok((0, 0));
|
||||
}
|
||||
|
||||
let mut wtxn = env.write_txn().map_err(|source| Error::DbStartWriteTxn {
|
||||
db: Self::LABEL,
|
||||
source,
|
||||
})?;
|
||||
|
||||
for key in &to_delete {
|
||||
db.delete(&mut wtxn, key).map_err(|source| Error::DbWrite {
|
||||
db: Self::LABEL,
|
||||
source,
|
||||
})?;
|
||||
}
|
||||
|
||||
for (key, accesses) in &to_update {
|
||||
db.put(&mut wtxn, key, accesses)
|
||||
.map_err(|source| Error::DbWrite {
|
||||
db: Self::LABEL,
|
||||
source,
|
||||
})?;
|
||||
}
|
||||
wtxn.commit().map_err(|source| Error::DbCommit {
|
||||
db: Self::LABEL,
|
||||
source,
|
||||
})?;
|
||||
|
||||
Ok((to_delete.len(), to_update.len()))
|
||||
}
|
||||
|
||||
fn get_accesses(&self, path: &Path) -> Result<Option<VecDeque<u64>>> {
|
||||
let key_hash = Self::path_to_hash_bytes(path)?;
|
||||
|
||||
let rtxn = self
|
||||
.env
|
||||
.read_txn()
|
||||
.map_err(|source| Error::DbStartReadTxn {
|
||||
db: Self::LABEL,
|
||||
source,
|
||||
})?;
|
||||
let result = self
|
||||
.db
|
||||
.get(&rtxn, &key_hash)
|
||||
.map_err(|source| Error::DbRead {
|
||||
db: Self::LABEL,
|
||||
source,
|
||||
})?;
|
||||
rtxn.commit().map_err(|source| Error::DbCommit {
|
||||
db: Self::LABEL,
|
||||
source,
|
||||
})?;
|
||||
|
||||
Ok(result)
|
||||
}
|
||||
|
||||
fn get_now(&self) -> u64 {
|
||||
SystemTime::now()
|
||||
.duration_since(UNIX_EPOCH)
|
||||
.unwrap()
|
||||
.as_secs()
|
||||
}
|
||||
|
||||
fn path_to_hash_bytes(path: &Path) -> Result<[u8; 32]> {
|
||||
let Some(key) = path.to_str() else {
|
||||
return Err(Error::InvalidPath(path.to_path_buf()));
|
||||
};
|
||||
|
||||
Ok(*blake3::hash(key.as_bytes()).as_bytes())
|
||||
}
|
||||
|
||||
/// Returns seconds since the most recent tracked access, or `None` if the
|
||||
/// file has never been tracked.
|
||||
pub fn seconds_since_last_access(&self, path: &Path) -> Result<Option<u64>> {
|
||||
let accesses = self.get_accesses(path)?;
|
||||
let last = accesses.and_then(|a| a.back().copied());
|
||||
Ok(last.map(|ts| self.get_now().saturating_sub(ts)))
|
||||
}
|
||||
|
||||
/// Number of tracked access for file path
|
||||
pub fn access_count(&self, path: &Path) -> Result<usize> {
|
||||
Ok(self.get_accesses(path)?.map_or(0, |a| a.len()))
|
||||
}
|
||||
|
||||
pub fn track_access(&self, path: &Path) -> Result<()> {
|
||||
let key_hash = Self::path_to_hash_bytes(path)?;
|
||||
let mut accesses = self.get_accesses(path)?.unwrap_or_default();
|
||||
|
||||
let now = self.get_now();
|
||||
let cutoff_time = now.saturating_sub((MAX_HISTORY_DAYS * SECONDS_PER_DAY) as u64);
|
||||
|
||||
// Drop stale timestamps from the front while also enforcing the
|
||||
// per-file cap. Reserves one slot for the `push_back` below.
|
||||
while let Some(&front_time) = accesses.front() {
|
||||
if front_time < cutoff_time || accesses.len() >= MAX_TIMESTAMPS_PER_FILE {
|
||||
accesses.pop_front();
|
||||
} else {
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
accesses.push_back(now);
|
||||
tracing::debug!(?path, accesses = accesses.len(), "Tracking access");
|
||||
|
||||
let mut wtxn = self
|
||||
.env
|
||||
.write_txn()
|
||||
.map_err(|source| Error::DbStartWriteTxn {
|
||||
db: Self::LABEL,
|
||||
source,
|
||||
})?;
|
||||
if let Err(e) = self.db.put(&mut wtxn, &key_hash, &accesses) {
|
||||
if is_map_full(&e) {
|
||||
self.health.mark_unhealthy("MDB_MAP_FULL on put");
|
||||
tracing::error!(
|
||||
?path,
|
||||
"Frecency DB hit MDB_MAP_FULL; dropping write — db will be \
|
||||
erased on next open via LmdbStore::erase_if_oversized"
|
||||
);
|
||||
return Ok(());
|
||||
}
|
||||
return Err(Error::DbWrite {
|
||||
db: Self::LABEL,
|
||||
source: e,
|
||||
});
|
||||
}
|
||||
|
||||
wtxn.commit()
|
||||
.inspect_err(|e| {
|
||||
if is_map_full(e) {
|
||||
self.health.mark_unhealthy("MDB_MAP_FULL on commit");
|
||||
tracing::error!(
|
||||
?path,
|
||||
"Frecency DB hit MDB_MAP_FULL on commit; dropping write"
|
||||
);
|
||||
}
|
||||
})
|
||||
.map_err(|source| Error::DbCommit {
|
||||
db: Self::LABEL,
|
||||
source,
|
||||
})
|
||||
}
|
||||
|
||||
pub fn get_access_score(&self, file_path: &Path, mode: FFFMode) -> i64 {
|
||||
let accesses = self
|
||||
.get_accesses(file_path)
|
||||
.ok()
|
||||
.flatten()
|
||||
.unwrap_or_default();
|
||||
|
||||
if accesses.is_empty() {
|
||||
return 0;
|
||||
}
|
||||
|
||||
let decay_constant = if mode.is_ai() {
|
||||
AI_DECAY_CONSTANT
|
||||
} else {
|
||||
DECAY_CONSTANT
|
||||
};
|
||||
let max_history_days = if mode.is_ai() {
|
||||
AI_MAX_HISTORY_DAYS
|
||||
} else {
|
||||
MAX_HISTORY_DAYS
|
||||
};
|
||||
|
||||
let now = self.get_now();
|
||||
let mut total_frecency = 0.0;
|
||||
|
||||
let cutoff_time = now.saturating_sub((max_history_days * SECONDS_PER_DAY) as u64);
|
||||
|
||||
for &access_time in accesses.iter().rev() {
|
||||
if access_time < cutoff_time {
|
||||
break; // All remaining entries are older, stop processing
|
||||
}
|
||||
|
||||
let days_ago = (now.saturating_sub(access_time) as f64) / SECONDS_PER_DAY;
|
||||
let decay_factor = (-decay_constant * days_ago).exp();
|
||||
total_frecency += decay_factor;
|
||||
}
|
||||
|
||||
let normalized_frecency = if total_frecency <= 10.0 {
|
||||
total_frecency
|
||||
} else {
|
||||
10.0 + (total_frecency - 10.0).sqrt() // Diminishing: >10 accesses grow slowly
|
||||
};
|
||||
|
||||
normalized_frecency.round() as i64
|
||||
}
|
||||
|
||||
/// Calculating modification score but only if the file is modified in the current git dir
|
||||
pub fn get_modification_score(
|
||||
&self,
|
||||
modified_time: u64,
|
||||
git_status: Option<git2::Status>,
|
||||
mode: FFFMode,
|
||||
) -> i64 {
|
||||
let is_modified_git_status = git_status.is_some_and(is_modified_status);
|
||||
if !is_modified_git_status {
|
||||
return 0;
|
||||
}
|
||||
|
||||
let thresholds = if mode.is_ai() {
|
||||
&AI_MODIFICATION_THRESHOLDS
|
||||
} else {
|
||||
&MODIFICATION_THRESHOLDS
|
||||
};
|
||||
|
||||
let now = self.get_now();
|
||||
let duration_since = now.saturating_sub(modified_time);
|
||||
|
||||
for i in 0..thresholds.len() {
|
||||
let (current_points, current_threshold) = thresholds[i];
|
||||
|
||||
if duration_since <= current_threshold {
|
||||
if i == 0 || duration_since == current_threshold {
|
||||
return current_points;
|
||||
}
|
||||
|
||||
let (prev_points, prev_threshold) = thresholds[i - 1];
|
||||
|
||||
let time_range = current_threshold - prev_threshold;
|
||||
let time_offset = duration_since - prev_threshold;
|
||||
let points_diff = prev_points - current_points;
|
||||
|
||||
let interpolated_score =
|
||||
prev_points - (points_diff * time_offset as i64) / time_range as i64;
|
||||
|
||||
return interpolated_score;
|
||||
}
|
||||
}
|
||||
|
||||
0
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::file_picker::FFFMode;
|
||||
|
||||
fn calculate_test_frecency_score(access_timestamps: &[u64], current_time: u64) -> i64 {
|
||||
let mut total_frecency = 0.0;
|
||||
|
||||
for &access_time in access_timestamps {
|
||||
let days_ago = (current_time.saturating_sub(access_time) as f64) / SECONDS_PER_DAY;
|
||||
let decay_factor = (-DECAY_CONSTANT * days_ago).exp();
|
||||
total_frecency += decay_factor;
|
||||
}
|
||||
|
||||
let normalized_frecency = if total_frecency <= 20.0 {
|
||||
total_frecency
|
||||
} else {
|
||||
20.0 + (total_frecency - 10.0).sqrt()
|
||||
};
|
||||
|
||||
normalized_frecency.round() as i64
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_frecency_calculation() {
|
||||
let current_time = 1000000000; // Base timestamp
|
||||
|
||||
let score = calculate_test_frecency_score(&[], current_time);
|
||||
assert_eq!(score, 0);
|
||||
|
||||
let accesses = [current_time]; // Accessed right now
|
||||
let score = calculate_test_frecency_score(&accesses, current_time);
|
||||
assert_eq!(score, 1); // 1.0 decay factor = 1
|
||||
|
||||
let ten_days_seconds = 10 * 86400; // 10 days in seconds
|
||||
let accesses = [current_time - ten_days_seconds];
|
||||
let score = calculate_test_frecency_score(&accesses, current_time);
|
||||
assert_eq!(score, 1); // ~0.5 decay factor rounds to 1
|
||||
|
||||
let accesses = [
|
||||
current_time, // Today
|
||||
current_time - 86400, // 1 day ago
|
||||
current_time - 172800, // 2 days ago
|
||||
];
|
||||
let score = calculate_test_frecency_score(&accesses, current_time);
|
||||
assert!(score > 2 && score < 4, "Score: {}", score); // About 3 accesses with decay
|
||||
|
||||
let thirty_days = 30 * 86400;
|
||||
let accesses = [current_time - thirty_days]; // 30 days ago
|
||||
let score = calculate_test_frecency_score(&accesses, current_time);
|
||||
assert!(
|
||||
score < 2,
|
||||
"Old access should have minimal score, got: {}",
|
||||
score
|
||||
);
|
||||
|
||||
let recent_frequent = [current_time, current_time - 86400, current_time - 172800];
|
||||
let old_single = [current_time - ten_days_seconds];
|
||||
|
||||
let recent_score = calculate_test_frecency_score(&recent_frequent, current_time);
|
||||
let old_score = calculate_test_frecency_score(&old_single, current_time);
|
||||
|
||||
assert!(
|
||||
recent_score > old_score,
|
||||
"Recent frequent access ({}) should score higher than old single access ({})",
|
||||
recent_score,
|
||||
old_score
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_modification_score_interpolation() {
|
||||
let temp_dir = std::env::temp_dir().join("fff_test_interpolation");
|
||||
let _ = std::fs::remove_dir_all(&temp_dir);
|
||||
let tracker = FrecencyTracker::open(temp_dir.to_str().unwrap()).unwrap();
|
||||
|
||||
let current_time = tracker.get_now();
|
||||
let git_status = Some(git2::Status::WT_MODIFIED);
|
||||
|
||||
// At 5 minutes: should interpolate between 16 and 8 points
|
||||
let five_minutes_ago = current_time - (5 * 60);
|
||||
let score = tracker.get_modification_score(five_minutes_ago, git_status, FFFMode::Neovim);
|
||||
|
||||
// Expected: 16 - (8 * 3 / 13) = 16 - 1 = 15 points
|
||||
// (time_offset = 5-2 = 3, time_range = 15-2 = 13, points_diff = 16-8 = 8)
|
||||
assert_eq!(score, 15, "5 minutes should interpolate to 15 points");
|
||||
|
||||
let two_minutes_ago = current_time - (2 * 60);
|
||||
let score = tracker.get_modification_score(two_minutes_ago, git_status, FFFMode::Neovim);
|
||||
assert_eq!(score, 16, "2 minutes should be exactly 16 points");
|
||||
|
||||
let fifteen_minutes_ago = current_time - (15 * 60);
|
||||
let score =
|
||||
tracker.get_modification_score(fifteen_minutes_ago, git_status, FFFMode::Neovim);
|
||||
assert_eq!(score, 8, "15 minutes should be exactly 8 points");
|
||||
|
||||
// At 12 hours: should interpolate between 4 and 2 points
|
||||
let twelve_hours_ago = current_time - (12 * 60 * 60);
|
||||
let score = tracker.get_modification_score(twelve_hours_ago, git_status, FFFMode::Neovim);
|
||||
// Expected: 4 - (2 * 11 / 23) = 4 - 0 = 4 points (integer division)
|
||||
// (time_offset = 12-1 = 11 hours, time_range = 24-1 = 23 hours, points_diff = 4-2 = 2)
|
||||
assert_eq!(score, 4, "12 hours should interpolate to 4 points");
|
||||
|
||||
// at 18 hours for more significant interpolation
|
||||
let eighteen_hours_ago = current_time - (18 * 60 * 60);
|
||||
let score = tracker.get_modification_score(eighteen_hours_ago, git_status, FFFMode::Neovim);
|
||||
// Expected: 4 - (2 * 17 / 23) = 4 - 1 = 3 points
|
||||
assert_eq!(score, 3, "18 hours should interpolate to 3 points");
|
||||
|
||||
let score = tracker.get_modification_score(five_minutes_ago, None, FFFMode::Neovim);
|
||||
assert_eq!(score, 0, "No git status should return 0");
|
||||
|
||||
let _ = std::fs::remove_dir_all(&temp_dir);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,259 @@
|
||||
use heed::{Database, Env, EnvOpenOptions};
|
||||
use std::fs;
|
||||
use std::path::Path;
|
||||
use std::sync::Arc;
|
||||
use std::sync::RwLock;
|
||||
use std::sync::atomic::{AtomicU8, Ordering};
|
||||
use std::thread;
|
||||
use std::time::Duration;
|
||||
|
||||
use crate::error::{Error, Result};
|
||||
|
||||
pub(crate) fn is_map_full(err: &heed::Error) -> bool {
|
||||
matches!(err, heed::Error::Mdb(heed::MdbError::MapFull))
|
||||
}
|
||||
|
||||
#[repr(u8)]
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
enum DbHealthState {
|
||||
Pending = 0,
|
||||
Healthy = 1,
|
||||
Degraded = 2,
|
||||
}
|
||||
|
||||
impl DbHealthState {
|
||||
fn from_u8(v: u8) -> Self {
|
||||
debug_assert!(v <= 2);
|
||||
|
||||
match v {
|
||||
0 => Self::Pending,
|
||||
1 => Self::Healthy,
|
||||
_ => Self::Degraded,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Default)]
|
||||
pub(crate) struct DbHealth(Arc<AtomicU8>);
|
||||
|
||||
impl DbHealth {
|
||||
pub(crate) fn new() -> Self {
|
||||
Self(Arc::new(AtomicU8::new(DbHealthState::Pending as u8)))
|
||||
}
|
||||
|
||||
pub(crate) fn is_healthy(&self) -> bool {
|
||||
// Pending counts as unhealthy: if the GC thread never flipped to
|
||||
// Healthy, something's wrong (deadlocked clear_stale_readers, stuck
|
||||
// writer mutex, etc.) and we want that surfaced to the user.
|
||||
DbHealthState::from_u8(self.0.load(Ordering::Acquire)) == DbHealthState::Healthy
|
||||
}
|
||||
|
||||
pub(crate) fn mark_healthy(&self) {
|
||||
let _ = self.0.compare_exchange(
|
||||
DbHealthState::Pending as u8,
|
||||
DbHealthState::Healthy as u8,
|
||||
Ordering::AcqRel,
|
||||
Ordering::Acquire,
|
||||
);
|
||||
}
|
||||
|
||||
pub(crate) fn mark_unhealthy(&self, reason: &'static str) {
|
||||
let prev = self.0.swap(DbHealthState::Degraded as u8, Ordering::AcqRel);
|
||||
if DbHealthState::from_u8(prev) != DbHealthState::Degraded {
|
||||
tracing::error!(reason, "LMDB tracker marked unhealthy");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Spawns a background thread that is ensuring that the environment that was previously
|
||||
/// open is safe, accessible and doesn't have a corrupted lock.md file. If it does this thread will
|
||||
/// hang indefinitely but we will have the information that the database is in failure mode
|
||||
pub(crate) fn spawn_lmdb_gc<T: LmdbStore>(shared: Arc<RwLock<Option<T>>>) {
|
||||
let thread_shared = shared.clone();
|
||||
let spawn_result = thread::Builder::new()
|
||||
.name("fff-lmdb-gc".into())
|
||||
.spawn(move || {
|
||||
// Holding a read guard blocks `destroy` / re-init's write
|
||||
// guard until this thread finishes — natural serialization.
|
||||
let guard = match thread_shared.read() {
|
||||
Ok(g) => g,
|
||||
Err(e) => {
|
||||
tracing::debug!("gc: read lock poisoned: {e}");
|
||||
return;
|
||||
}
|
||||
};
|
||||
let Some(ref tracker) = *guard else {
|
||||
return; // destroyed before we started
|
||||
};
|
||||
let env = tracker.env();
|
||||
|
||||
if let Err(e) = T::purge_stale_data(env) {
|
||||
tracing::debug!("purge_stale_data failed: {e}");
|
||||
}
|
||||
|
||||
tracker.health().mark_healthy();
|
||||
});
|
||||
|
||||
if let Err(e) = spawn_result {
|
||||
tracing::debug!(?e, "failed to spawn fff-lmdb-gc thread");
|
||||
// No thread = mark healthy now so healthcheck isn't stuck Pending.
|
||||
if let Ok(guard) = shared.read()
|
||||
&& let Some(ref tracker) = *guard
|
||||
{
|
||||
tracker.health().mark_healthy();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Concurrent `mdb_env_open` calls on the same path can race on macOS
|
||||
// this is for some reason fixabtly by simple retry of the open
|
||||
fn is_transient_env_open_error(err: &heed::Error) -> bool {
|
||||
match err {
|
||||
heed::Error::Io(io) => matches!(
|
||||
io.kind(),
|
||||
std::io::ErrorKind::InvalidInput | std::io::ErrorKind::NotFound
|
||||
),
|
||||
_ => false,
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) trait LmdbStore: Sized + Send + Sync + 'static {
|
||||
/// Short label used to defferintiate different instances of this trait
|
||||
const LABEL: &'static str;
|
||||
/// LMDB map size in bytes. Must be a multiple of the OS page size.
|
||||
const MAP_SIZE: usize;
|
||||
/// Number of named sub-databases. `0` for single-db envs.
|
||||
const MAX_DBS: u32;
|
||||
/// Hard cap on `data.mdb` size.
|
||||
const SIZE_CAP_BYTES: u64;
|
||||
|
||||
/// Borrow the env in the read lock
|
||||
fn env(&self) -> &Env;
|
||||
/// Borrow the health flag from the tracker.
|
||||
fn health(&self) -> &DbHealth;
|
||||
|
||||
/// Override to purge stale rows, compact, etc. Default no-op. Runs on
|
||||
/// the GC thread while a read lock is held against the shared handle,
|
||||
/// so destroy / re-init naturally wait for it.
|
||||
fn purge_stale_data(_env: &Env) -> Result<()> {
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Open the LMDB env. Returns env + a `DbHealth` starting in Pending;
|
||||
/// the GC thread spawned by `spawn_gc` flips it to Healthy. Write
|
||||
/// paths flip it to Degraded on MDB_MAP_FULL.
|
||||
#[tracing::instrument]
|
||||
fn open_env(db_path: &Path) -> Result<(Env, DbHealth)> {
|
||||
Self::erase_if_oversized(db_path);
|
||||
fs::create_dir_all(db_path).map_err(Error::CreateDir)?;
|
||||
let db = Self::LABEL;
|
||||
|
||||
const MAX_ATTEMPTS: u32 = 8;
|
||||
let mut attempt = 0u32;
|
||||
let env = loop {
|
||||
let result = unsafe {
|
||||
let mut opts = EnvOpenOptions::new();
|
||||
opts.map_size(Self::MAP_SIZE);
|
||||
if Self::MAX_DBS > 0 {
|
||||
opts.max_dbs(Self::MAX_DBS);
|
||||
}
|
||||
opts.open(db_path)
|
||||
};
|
||||
|
||||
match result {
|
||||
Ok(env) => break env,
|
||||
Err(e) if is_transient_env_open_error(&e) && attempt + 1 < MAX_ATTEMPTS => {
|
||||
attempt += 1;
|
||||
tracing::debug!(
|
||||
path = %db_path.display(),
|
||||
attempt,
|
||||
error = ?e,
|
||||
"transient LMDB env open error, retrying"
|
||||
);
|
||||
|
||||
thread::sleep(Duration::from_millis(50));
|
||||
}
|
||||
Err(e) => return Err(Error::EnvOpen { db, source: e }),
|
||||
}
|
||||
};
|
||||
|
||||
// Reclaim reader slots left behind by prior processes that died
|
||||
// without cleanup. Must run before we start any read txns (which
|
||||
// open_database_safe does) — otherwise we may hit MDB_READERS_FULL
|
||||
// on a fresh env just because lock.mdb still has stale entries
|
||||
// from a previous crash.
|
||||
//
|
||||
// This is the one LMDB maintenance call we run on the caller's
|
||||
// thread. If the lock file is genuinely wedged this will block
|
||||
// forever, but the alternative — never getting past init — is
|
||||
// worse and the bg-thread trick doesn't solve it anyway.
|
||||
match env.clear_stale_readers() {
|
||||
Ok(cleared) if cleared > 0 => {
|
||||
tracing::warn!(cleared, "reclaimed stale LMDB reader slots at open");
|
||||
}
|
||||
Ok(_) => {}
|
||||
Err(e) => tracing::debug!("clear_stale_readers at open failed: {e}"),
|
||||
}
|
||||
|
||||
Ok((env, DbHealth::new()))
|
||||
}
|
||||
|
||||
/// Open or create a database without blocking on the LMDB writer mutex
|
||||
/// when the database already exists.
|
||||
fn open_database_safe<KC, DC>(env: &Env, name: Option<&str>) -> Result<Database<KC, DC>>
|
||||
where
|
||||
KC: 'static,
|
||||
DC: 'static,
|
||||
{
|
||||
let db = Self::LABEL;
|
||||
let rtxn = env
|
||||
.read_txn()
|
||||
.map_err(|source| Error::DbStartReadTxn { db, source })?;
|
||||
let maybe_db: Option<Database<KC, DC>> = env
|
||||
.open_database(&rtxn, name)
|
||||
.map_err(|source| Error::DbOpen { db, source })?;
|
||||
|
||||
// do not drop the DB here
|
||||
rtxn.commit()
|
||||
.map_err(|source| Error::DbCommit { db, source })?;
|
||||
|
||||
match maybe_db {
|
||||
Some(handle) => Ok(handle),
|
||||
None => {
|
||||
// First time: create the database (requires write lock).
|
||||
// unfortunately this CAN be deadlocking and this is what we see happens
|
||||
// if the other part of the code is segfaulting, so the only rule to prevent this
|
||||
// write the good code mf, okay?
|
||||
let mut wtxn = env
|
||||
.write_txn()
|
||||
.map_err(|source| Error::DbStartWriteTxn { db, source })?;
|
||||
let handle = env
|
||||
.create_database(&mut wtxn, name)
|
||||
.map_err(|source| Error::DbCreate { db, source })?;
|
||||
|
||||
wtxn.commit()
|
||||
.map_err(|source| Error::DbCommit { db, source })?;
|
||||
Ok(handle)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn erase_if_oversized(db_path: &Path) {
|
||||
let data = db_path.join("data.mdb");
|
||||
let Ok(meta) = fs::metadata(&data) else {
|
||||
return;
|
||||
};
|
||||
if meta.len() <= Self::SIZE_CAP_BYTES {
|
||||
return;
|
||||
}
|
||||
|
||||
tracing::error!(
|
||||
path = %db_path.display(),
|
||||
size = meta.len(),
|
||||
cap = Self::SIZE_CAP_BYTES,
|
||||
"LMDB db exceeds size cap, erasing"
|
||||
);
|
||||
let _ = fs::remove_file(&data);
|
||||
let _ = fs::remove_file(db_path.join("lock.mdb"));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,4 @@
|
||||
pub mod db_healthcheck;
|
||||
pub mod frecency;
|
||||
pub(crate) mod lmdb;
|
||||
pub mod query_tracker;
|
||||
@@ -1,11 +1,10 @@
|
||||
use crate::db_healthcheck::DbHealthChecker;
|
||||
use super::db_healthcheck::DbHealthChecker;
|
||||
use super::lmdb::{DbHealth, LmdbStore, is_map_full};
|
||||
use crate::error::Error;
|
||||
use heed::types::Bytes;
|
||||
use heed::{Database, Env, EnvOpenOptions};
|
||||
use heed::{EnvFlags, types::SerdeBincode};
|
||||
use heed::types::{Bytes, SerdeBincode};
|
||||
use heed::{Database, Env};
|
||||
use serde::{Deserialize, Serialize};
|
||||
use std::collections::VecDeque;
|
||||
use std::fs;
|
||||
use std::path::{Path, PathBuf};
|
||||
use std::time::{SystemTime, UNIX_EPOCH};
|
||||
|
||||
@@ -35,6 +34,7 @@ pub struct QueryTracker {
|
||||
query_history_db: Database<Bytes, SerdeBincode<VecDeque<HistoryEntry>>>,
|
||||
// Database for project_path -> VecDeque<HistoryEntry> mappings (grep)
|
||||
grep_query_history_db: Database<Bytes, SerdeBincode<VecDeque<HistoryEntry>>>,
|
||||
health: DbHealth,
|
||||
}
|
||||
|
||||
impl DbHealthChecker for QueryTracker {
|
||||
@@ -42,15 +42,40 @@ impl DbHealthChecker for QueryTracker {
|
||||
&self.env
|
||||
}
|
||||
|
||||
fn count_entries(&self) -> Result<Vec<(&'static str, u64)>, Error> {
|
||||
let rtxn = self.env.read_txn().map_err(Error::DbStartReadTxn)?;
|
||||
fn is_healthy(&self) -> bool {
|
||||
self.health.is_healthy()
|
||||
}
|
||||
|
||||
let count_queries = self.query_file_db.len(&rtxn).map_err(Error::DbRead)?;
|
||||
let count_histories = self.query_history_db.len(&rtxn).map_err(Error::DbRead)?;
|
||||
let count_grep_histories = self
|
||||
.grep_query_history_db
|
||||
fn count_entries(&self) -> Result<Vec<(&'static str, u64)>, Error> {
|
||||
let rtxn = self
|
||||
.env
|
||||
.read_txn()
|
||||
.map_err(|source| Error::DbStartReadTxn {
|
||||
db: Self::LABEL,
|
||||
source,
|
||||
})?;
|
||||
|
||||
let count_queries = self
|
||||
.query_file_db
|
||||
.len(&rtxn)
|
||||
.map_err(Error::DbRead)?;
|
||||
.map_err(|source| Error::DbRead {
|
||||
db: Self::LABEL,
|
||||
source,
|
||||
})?;
|
||||
let count_histories = self
|
||||
.query_history_db
|
||||
.len(&rtxn)
|
||||
.map_err(|source| Error::DbRead {
|
||||
db: Self::LABEL,
|
||||
source,
|
||||
})?;
|
||||
let count_grep_histories =
|
||||
self.grep_query_history_db
|
||||
.len(&rtxn)
|
||||
.map_err(|source| Error::DbRead {
|
||||
db: Self::LABEL,
|
||||
source,
|
||||
})?;
|
||||
|
||||
Ok(vec![
|
||||
("query_file_entries", count_queries),
|
||||
@@ -60,44 +85,54 @@ impl DbHealthChecker for QueryTracker {
|
||||
}
|
||||
}
|
||||
|
||||
impl LmdbStore for QueryTracker {
|
||||
const LABEL: &'static str = "query";
|
||||
// 10 MiB hard ceiling. Same reasoning as FrecencyTracker (GH issue #437).
|
||||
const MAP_SIZE: usize = 10 * 1024 * 1024;
|
||||
const MAX_DBS: u32 = 16;
|
||||
const SIZE_CAP_BYTES: u64 = 8 * 1024 * 1024;
|
||||
|
||||
fn env(&self) -> &Env {
|
||||
&self.env
|
||||
}
|
||||
|
||||
fn health(&self) -> &DbHealth {
|
||||
&self.health
|
||||
}
|
||||
}
|
||||
|
||||
impl QueryTracker {
|
||||
pub fn new(db_path: &str, use_unsafe_no_lock: bool) -> Result<Self, Error> {
|
||||
fs::create_dir_all(db_path).map_err(Error::CreateDir)?;
|
||||
let env = unsafe {
|
||||
let mut opts = EnvOpenOptions::new();
|
||||
opts.max_dbs(16); // Allow up to 16 databases per environment
|
||||
if use_unsafe_no_lock {
|
||||
opts.flags(EnvFlags::NO_LOCK | EnvFlags::NO_SYNC | EnvFlags::NO_META_SYNC);
|
||||
}
|
||||
opts.open(db_path).map_err(Error::EnvOpen)?
|
||||
};
|
||||
/// Returns the on-disk path of the LMDB environment directory.
|
||||
pub fn db_path(&self) -> &Path {
|
||||
self.env.path()
|
||||
}
|
||||
|
||||
env.clear_stale_readers()
|
||||
.map_err(Error::DbClearStaleReaders)?;
|
||||
pub fn open(db_path: impl AsRef<Path>) -> Result<Self, Error> {
|
||||
let db_path = db_path.as_ref();
|
||||
let (env, health) = Self::open_env(db_path)?;
|
||||
|
||||
let mut wtxn = env.write_txn().map_err(Error::DbStartWriteTxn)?;
|
||||
|
||||
// Create two named databases
|
||||
let query_file_db = env
|
||||
.create_database(&mut wtxn, Some("query_file_associations"))
|
||||
.map_err(Error::DbCreate)?;
|
||||
let query_history_db = env
|
||||
.create_database(&mut wtxn, Some("query_history"))
|
||||
.map_err(Error::DbCreate)?;
|
||||
let grep_query_history_db = env
|
||||
.create_database(&mut wtxn, Some("grep_query_history"))
|
||||
.map_err(Error::DbCreate)?;
|
||||
|
||||
wtxn.commit().map_err(Error::DbCommit)?;
|
||||
let query_file_db = Self::open_database_safe(&env, Some("query_file_associations"))?;
|
||||
let query_history_db = Self::open_database_safe(&env, Some("query_history"))?;
|
||||
let grep_query_history_db = Self::open_database_safe(&env, Some("grep_query_history"))?;
|
||||
|
||||
Ok(QueryTracker {
|
||||
env,
|
||||
query_file_db,
|
||||
query_history_db,
|
||||
grep_query_history_db,
|
||||
health,
|
||||
})
|
||||
}
|
||||
|
||||
#[deprecated(
|
||||
since = "0.7.0",
|
||||
note = "LMDB unsafe no-lock mode is no longer supported; use `QueryTracker::open` instead. \
|
||||
The `_use_unsafe_no_lock` argument is ignored."
|
||||
)]
|
||||
pub fn new(db_path: impl AsRef<Path>, _use_unsafe_no_lock: bool) -> Result<Self, Error> {
|
||||
Self::open(db_path)
|
||||
}
|
||||
|
||||
fn get_now(&self) -> u64 {
|
||||
SystemTime::now()
|
||||
.duration_since(UNIX_EPOCH)
|
||||
@@ -136,7 +171,10 @@ impl QueryTracker {
|
||||
) -> Result<(), Error> {
|
||||
let mut history = db
|
||||
.get(wtxn, project_key)
|
||||
.map_err(Error::DbRead)?
|
||||
.map_err(|source| Error::DbRead {
|
||||
db: Self::LABEL,
|
||||
source,
|
||||
})?
|
||||
.unwrap_or_default();
|
||||
|
||||
history.push_back(HistoryEntry {
|
||||
@@ -148,7 +186,10 @@ impl QueryTracker {
|
||||
}
|
||||
|
||||
db.put(wtxn, project_key, &history)
|
||||
.map_err(Error::DbWrite)?;
|
||||
.map_err(|source| Error::DbWrite {
|
||||
db: Self::LABEL,
|
||||
source,
|
||||
})?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
@@ -160,11 +201,17 @@ impl QueryTracker {
|
||||
project_key: &[u8; 32],
|
||||
offset: usize,
|
||||
) -> Result<Option<String>, Error> {
|
||||
let rtxn = env.read_txn().map_err(Error::DbStartReadTxn)?;
|
||||
let rtxn = env.read_txn().map_err(|source| Error::DbStartReadTxn {
|
||||
db: Self::LABEL,
|
||||
source,
|
||||
})?;
|
||||
|
||||
let mut history = db
|
||||
.get(&rtxn, project_key)
|
||||
.map_err(Error::DbRead)?
|
||||
.map_err(|source| Error::DbRead {
|
||||
db: Self::LABEL,
|
||||
source,
|
||||
})?
|
||||
.unwrap_or_default();
|
||||
|
||||
// history is FIFO, last element is most recent
|
||||
@@ -187,12 +234,21 @@ impl QueryTracker {
|
||||
let file_path_buf = file_path.to_path_buf();
|
||||
|
||||
let query_key = Self::create_query_key(project_path, query)?;
|
||||
let mut wtxn = self.env.write_txn().map_err(Error::DbStartWriteTxn)?;
|
||||
let mut wtxn = self
|
||||
.env
|
||||
.write_txn()
|
||||
.map_err(|source| Error::DbStartWriteTxn {
|
||||
db: Self::LABEL,
|
||||
source,
|
||||
})?;
|
||||
|
||||
let mut entry = self
|
||||
.query_file_db
|
||||
.get(&wtxn, &query_key)
|
||||
.map_err(Error::DbRead)?
|
||||
.map_err(|source| Error::DbRead {
|
||||
db: Self::LABEL,
|
||||
source,
|
||||
})?
|
||||
.unwrap_or_else(|| QueryMatchEntry {
|
||||
file_path: file_path_buf.clone(),
|
||||
open_count: 0,
|
||||
@@ -222,15 +278,50 @@ impl QueryTracker {
|
||||
|
||||
entry.last_opened = now;
|
||||
|
||||
self.query_file_db
|
||||
.put(&mut wtxn, &query_key, &entry)
|
||||
.map_err(Error::DbWrite)?;
|
||||
if let Err(e) = self.query_file_db.put(&mut wtxn, &query_key, &entry) {
|
||||
if is_map_full(&e) {
|
||||
self.health.mark_unhealthy("MDB_MAP_FULL on put");
|
||||
tracing::error!(
|
||||
?query,
|
||||
"Query tracker DB hit MDB_MAP_FULL; dropping write — db will \
|
||||
be erased on next open"
|
||||
);
|
||||
return Ok(());
|
||||
}
|
||||
return Err(Error::DbWrite {
|
||||
db: Self::LABEL,
|
||||
source: e,
|
||||
});
|
||||
}
|
||||
|
||||
// Update query history database
|
||||
let project_key = Self::create_project_key(project_path)?;
|
||||
Self::append_to_history(&self.query_history_db, &mut wtxn, &project_key, query, now)?;
|
||||
if let Err(e) =
|
||||
Self::append_to_history(&self.query_history_db, &mut wtxn, &project_key, query, now)
|
||||
{
|
||||
if let Error::DbWrite {
|
||||
source: ref inner, ..
|
||||
} = e
|
||||
&& is_map_full(inner)
|
||||
{
|
||||
self.health.mark_unhealthy("MDB_MAP_FULL on history append");
|
||||
tracing::error!(?query, "Query tracker DB map full while appending history");
|
||||
return Ok(());
|
||||
}
|
||||
return Err(e);
|
||||
}
|
||||
|
||||
wtxn.commit().map_err(Error::DbCommit)?;
|
||||
if let Err(e) = wtxn.commit() {
|
||||
if is_map_full(&e) {
|
||||
self.health.mark_unhealthy("MDB_MAP_FULL on commit");
|
||||
tracing::error!(?query, "Query tracker DB map full on commit");
|
||||
return Ok(());
|
||||
}
|
||||
return Err(Error::DbCommit {
|
||||
db: Self::LABEL,
|
||||
source: e,
|
||||
});
|
||||
}
|
||||
|
||||
tracing::debug!(?query, ?file_path, "Tracked query completion");
|
||||
Ok(())
|
||||
@@ -243,13 +334,21 @@ impl QueryTracker {
|
||||
min_combo_count: u32,
|
||||
) -> Result<Option<QueryMatchEntry>, Error> {
|
||||
let query_key = Self::create_query_key(project_path, query)?;
|
||||
tracing::debug!(?query_key, "HASH");
|
||||
let rtxn = self.env.read_txn().map_err(Error::DbStartReadTxn)?;
|
||||
let rtxn = self
|
||||
.env
|
||||
.read_txn()
|
||||
.map_err(|source| Error::DbStartReadTxn {
|
||||
db: Self::LABEL,
|
||||
source,
|
||||
})?;
|
||||
|
||||
let last_match = self
|
||||
.query_file_db
|
||||
.get(&rtxn, &query_key)
|
||||
.map_err(Error::DbRead)?;
|
||||
.map_err(|source| Error::DbRead {
|
||||
db: Self::LABEL,
|
||||
source,
|
||||
})?;
|
||||
|
||||
Ok(last_match.filter(|entry| entry.open_count >= min_combo_count))
|
||||
}
|
||||
@@ -263,13 +362,21 @@ impl QueryTracker {
|
||||
) -> Result<i32, Error> {
|
||||
let query_key = Self::create_query_key(project_path, query)?;
|
||||
tracing::debug!(?query_key, "HASH");
|
||||
let rtxn = self.env.read_txn().map_err(Error::DbStartReadTxn)?;
|
||||
let rtxn = self
|
||||
.env
|
||||
.read_txn()
|
||||
.map_err(|source| Error::DbStartReadTxn {
|
||||
db: Self::LABEL,
|
||||
source,
|
||||
})?;
|
||||
|
||||
match self
|
||||
.query_file_db
|
||||
.get(&rtxn, &query_key)
|
||||
.map_err(Error::DbRead)?
|
||||
{
|
||||
.map_err(|source| Error::DbRead {
|
||||
db: Self::LABEL,
|
||||
source,
|
||||
})? {
|
||||
Some(entry) => {
|
||||
// Check if the file path matches and return boost
|
||||
if entry.file_path == file_path && entry.open_count >= 2 {
|
||||
@@ -298,17 +405,45 @@ impl QueryTracker {
|
||||
pub fn track_grep_query(&mut self, query: &str, project_path: &Path) -> Result<(), Error> {
|
||||
let now = self.get_now();
|
||||
let project_key = Self::create_project_key(project_path)?;
|
||||
let mut wtxn = self.env.write_txn().map_err(Error::DbStartWriteTxn)?;
|
||||
let mut wtxn = self
|
||||
.env
|
||||
.write_txn()
|
||||
.map_err(|source| Error::DbStartWriteTxn {
|
||||
db: Self::LABEL,
|
||||
source,
|
||||
})?;
|
||||
|
||||
Self::append_to_history(
|
||||
if let Err(e) = Self::append_to_history(
|
||||
&self.grep_query_history_db,
|
||||
&mut wtxn,
|
||||
&project_key,
|
||||
query,
|
||||
now,
|
||||
)?;
|
||||
) {
|
||||
if let Error::DbWrite {
|
||||
source: ref inner, ..
|
||||
} = e
|
||||
&& is_map_full(inner)
|
||||
{
|
||||
self.health
|
||||
.mark_unhealthy("MDB_MAP_FULL on grep history append");
|
||||
tracing::error!(?query, "Grep query history DB map full; dropping write");
|
||||
return Ok(());
|
||||
}
|
||||
return Err(e);
|
||||
}
|
||||
|
||||
wtxn.commit().map_err(Error::DbCommit)?;
|
||||
if let Err(e) = wtxn.commit() {
|
||||
if is_map_full(&e) {
|
||||
self.health.mark_unhealthy("MDB_MAP_FULL on commit");
|
||||
tracing::error!(?query, "Grep query history DB map full on commit");
|
||||
return Ok(());
|
||||
}
|
||||
return Err(Error::DbCommit {
|
||||
db: Self::LABEL,
|
||||
source: e,
|
||||
});
|
||||
}
|
||||
|
||||
tracing::debug!(?query, "Tracked grep query");
|
||||
Ok(())
|
||||
@@ -336,7 +471,7 @@ mod tests {
|
||||
let temp_dir = env::temp_dir().join("fff_test_query_tracking_new");
|
||||
let _ = std::fs::remove_dir_all(&temp_dir);
|
||||
|
||||
let mut tracker = QueryTracker::new(temp_dir.to_str().unwrap(), true).unwrap();
|
||||
let mut tracker = QueryTracker::open(temp_dir.to_str().unwrap()).unwrap();
|
||||
|
||||
let project_path = PathBuf::from("/test/project");
|
||||
let file_path = PathBuf::from("/test/project/src/main.rs");
|
||||
@@ -7,6 +7,10 @@ pub enum Error {
|
||||
ThreadPanic,
|
||||
#[error("Invalid path {0}")]
|
||||
InvalidPath(std::path::PathBuf),
|
||||
#[error(
|
||||
"Can not run certain FFF features in a file system root or home directories. Consider smaller per-project directories."
|
||||
)]
|
||||
FilesystemRoot(std::path::PathBuf),
|
||||
#[error("File picker not initialized")]
|
||||
FilePickerMissing,
|
||||
#[error("Failed to acquire lock for frecency")]
|
||||
@@ -17,24 +21,68 @@ pub enum Error {
|
||||
AcquirePathCacheLock,
|
||||
#[error("Failed to create directory: {0}")]
|
||||
CreateDir(#[from] std::io::Error),
|
||||
#[error("Failed to open frecency database env: {0}")]
|
||||
EnvOpen(#[source] heed::Error),
|
||||
#[error("Failed to create frecency database: {0}")]
|
||||
DbCreate(#[source] heed::Error),
|
||||
#[error("Failed to clear stale readers for frecency database: {0}")]
|
||||
DbClearStaleReaders(#[source] heed::Error),
|
||||
#[error("Failed to remove database directory {path}: {source}")]
|
||||
RemoveDbDir {
|
||||
path: std::path::PathBuf,
|
||||
source: std::io::Error,
|
||||
},
|
||||
#[error("Something is wrong with the local db instance: {0}")]
|
||||
GenericDbError(#[from] heed::Error),
|
||||
#[error("Failed to open {db} database env: {source}")]
|
||||
EnvOpen {
|
||||
db: &'static str,
|
||||
#[source]
|
||||
source: heed::Error,
|
||||
},
|
||||
#[error("Failed to create {db} database: {source}")]
|
||||
DbCreate {
|
||||
db: &'static str,
|
||||
#[source]
|
||||
source: heed::Error,
|
||||
},
|
||||
#[error("Failed to open {db} database: {source}")]
|
||||
DbOpen {
|
||||
db: &'static str,
|
||||
#[source]
|
||||
source: heed::Error,
|
||||
},
|
||||
#[error("Failed to clear stale readers for {db} database: {source}")]
|
||||
DbClearStaleReaders {
|
||||
db: &'static str,
|
||||
#[source]
|
||||
source: heed::Error,
|
||||
},
|
||||
|
||||
#[error("Failed to start read transaction for frecency database: {0}")]
|
||||
DbStartReadTxn(#[source] heed::Error),
|
||||
#[error("Failed to start write transaction for frecency database: {0}")]
|
||||
DbStartWriteTxn(#[source] heed::Error),
|
||||
|
||||
#[error("Failed to read from frecency database: {0}")]
|
||||
DbRead(#[source] heed::Error),
|
||||
#[error("Failed to write to frecency database: {0}")]
|
||||
DbWrite(#[source] heed::Error),
|
||||
#[error("Failed to commit write transaction to frecency database: {0}")]
|
||||
DbCommit(#[source] heed::Error),
|
||||
#[error("Failed to start read transaction for {db} database: {source}")]
|
||||
DbStartReadTxn {
|
||||
db: &'static str,
|
||||
#[source]
|
||||
source: heed::Error,
|
||||
},
|
||||
#[error("Failed to start write transaction for {db} database: {source}")]
|
||||
DbStartWriteTxn {
|
||||
db: &'static str,
|
||||
#[source]
|
||||
source: heed::Error,
|
||||
},
|
||||
#[error("Failed to read from {db} database: {source}")]
|
||||
DbRead {
|
||||
db: &'static str,
|
||||
#[source]
|
||||
source: heed::Error,
|
||||
},
|
||||
#[error("Failed to write to {db} database: {source}")]
|
||||
DbWrite {
|
||||
db: &'static str,
|
||||
#[source]
|
||||
source: heed::Error,
|
||||
},
|
||||
#[error("Failed to commit write transaction to {db} database: {source}")]
|
||||
DbCommit {
|
||||
db: &'static str,
|
||||
#[source]
|
||||
source: heed::Error,
|
||||
},
|
||||
#[error("Failed to start file system watcher: {0}")]
|
||||
FileSystemWatch(#[from] notify::Error),
|
||||
|
||||
|
||||
+1936
-430
File diff suppressed because it is too large
Load Diff
@@ -1,304 +0,0 @@
|
||||
use crate::db_healthcheck::DbHealthChecker;
|
||||
use crate::{error::Error, git::is_modified_status};
|
||||
use heed::{Database, Env, EnvOpenOptions};
|
||||
use heed::{
|
||||
EnvFlags,
|
||||
types::{Bytes, SerdeBincode},
|
||||
};
|
||||
use std::fs;
|
||||
use std::time::{SystemTime, UNIX_EPOCH};
|
||||
use std::{collections::VecDeque, path::Path};
|
||||
|
||||
const DECAY_CONSTANT: f64 = 0.0693; // ln(2)/10 for 10-day half-life
|
||||
const SECONDS_PER_DAY: f64 = 86400.0;
|
||||
const MAX_HISTORY_DAYS: f64 = 30.0; // Only consider accesses within 30 days
|
||||
|
||||
#[derive(Debug)]
|
||||
pub struct FrecencyTracker {
|
||||
env: Env,
|
||||
db: Database<Bytes, SerdeBincode<VecDeque<u64>>>,
|
||||
}
|
||||
|
||||
const MODIFICATION_THRESHOLDS: [(i64, u64); 5] = [
|
||||
(16, 60 * 2), // 2 minutes
|
||||
(8, 60 * 15), // 15 minutes
|
||||
(4, 60 * 60), // 1 hour
|
||||
(2, 60 * 60 * 24), // 1 day
|
||||
(1, 60 * 60 * 24 * 7), // 1 week
|
||||
];
|
||||
|
||||
impl DbHealthChecker for FrecencyTracker {
|
||||
fn get_env(&self) -> &heed::Env {
|
||||
&self.env
|
||||
}
|
||||
|
||||
fn count_entries(&self) -> Result<Vec<(&'static str, u64)>, Error> {
|
||||
let rtxn = self.env.read_txn().map_err(Error::DbStartReadTxn)?;
|
||||
let count = self.db.len(&rtxn).map_err(Error::DbRead)?;
|
||||
|
||||
Ok(vec![("absolute_frecency_entries", count)])
|
||||
}
|
||||
}
|
||||
|
||||
impl FrecencyTracker {
|
||||
pub fn new(db_path: &str, use_unsafe_no_lock: bool) -> Result<Self, Error> {
|
||||
fs::create_dir_all(db_path).map_err(Error::CreateDir)?;
|
||||
let env = unsafe {
|
||||
let mut opts = EnvOpenOptions::new();
|
||||
if use_unsafe_no_lock {
|
||||
opts.flags(EnvFlags::NO_LOCK | EnvFlags::NO_SYNC | EnvFlags::NO_META_SYNC);
|
||||
}
|
||||
opts.open(db_path).map_err(Error::EnvOpen)?
|
||||
};
|
||||
env.clear_stale_readers()
|
||||
.map_err(Error::DbClearStaleReaders)?;
|
||||
|
||||
// we will open the default unnamed database
|
||||
let mut wtxn = env.write_txn().map_err(Error::DbStartWriteTxn)?;
|
||||
let db = env
|
||||
.create_database(&mut wtxn, None)
|
||||
.map_err(Error::DbCreate)?;
|
||||
|
||||
Ok(FrecencyTracker {
|
||||
db,
|
||||
env: env.clone(),
|
||||
})
|
||||
}
|
||||
|
||||
fn get_accesses(&self, path: &Path) -> Result<Option<VecDeque<u64>>, Error> {
|
||||
let rtxn = self.env.read_txn().map_err(Error::DbStartReadTxn)?;
|
||||
|
||||
let key_hash = Self::path_to_hash_bytes(path)?;
|
||||
self.db.get(&rtxn, &key_hash).map_err(Error::DbRead)
|
||||
}
|
||||
|
||||
fn get_now(&self) -> u64 {
|
||||
SystemTime::now()
|
||||
.duration_since(UNIX_EPOCH)
|
||||
.unwrap()
|
||||
.as_secs()
|
||||
}
|
||||
|
||||
fn path_to_hash_bytes(path: &Path) -> Result<[u8; 32], Error> {
|
||||
let Some(key) = path.to_str() else {
|
||||
return Err(Error::InvalidPath(path.to_path_buf()));
|
||||
};
|
||||
|
||||
Ok(*blake3::hash(key.as_bytes()).as_bytes())
|
||||
}
|
||||
|
||||
pub fn track_access(&self, path: &Path) -> Result<(), Error> {
|
||||
let mut wtxn = self.env.write_txn().map_err(Error::DbStartWriteTxn)?;
|
||||
|
||||
let key_hash = Self::path_to_hash_bytes(path)?;
|
||||
let mut accesses = self.get_accesses(path)?.unwrap_or_default();
|
||||
|
||||
let now = self.get_now();
|
||||
let cutoff_time = now.saturating_sub((MAX_HISTORY_DAYS * SECONDS_PER_DAY) as u64);
|
||||
while let Some(&front_time) = accesses.front() {
|
||||
if front_time < cutoff_time {
|
||||
accesses.pop_front();
|
||||
} else {
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
accesses.push_back(now);
|
||||
tracing::debug!(?path, accesses = accesses.len(), "Tracking access");
|
||||
|
||||
self.db
|
||||
.put(&mut wtxn, &key_hash, &accesses)
|
||||
.map_err(Error::DbWrite)?;
|
||||
|
||||
wtxn.commit().map_err(Error::DbCommit)?;
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub fn get_access_score(&self, file_path: &Path) -> i64 {
|
||||
let accesses = self
|
||||
.get_accesses(file_path)
|
||||
.ok()
|
||||
.flatten()
|
||||
.unwrap_or_default();
|
||||
|
||||
if accesses.is_empty() {
|
||||
return 0;
|
||||
}
|
||||
|
||||
let now = self.get_now();
|
||||
let mut total_frecency = 0.0;
|
||||
|
||||
let cutoff_time = now.saturating_sub((MAX_HISTORY_DAYS * SECONDS_PER_DAY) as u64);
|
||||
|
||||
for &access_time in accesses.iter().rev() {
|
||||
if access_time < cutoff_time {
|
||||
break; // All remaining entries are older, stop processing
|
||||
}
|
||||
|
||||
let days_ago = (now.saturating_sub(access_time) as f64) / SECONDS_PER_DAY;
|
||||
let decay_factor = (-DECAY_CONSTANT * days_ago).exp();
|
||||
total_frecency += decay_factor;
|
||||
}
|
||||
|
||||
let normalized_frecency = if total_frecency <= 10.0 {
|
||||
total_frecency
|
||||
} else {
|
||||
10.0 + (total_frecency - 10.0).sqrt() // Diminishing: >10 accesses grow slowly
|
||||
};
|
||||
|
||||
normalized_frecency.round() as i64
|
||||
}
|
||||
|
||||
/// Calculating modification score but only if the file is modified in the current git dir
|
||||
pub fn get_modification_score(
|
||||
&self,
|
||||
modified_time: u64,
|
||||
git_status: Option<git2::Status>,
|
||||
) -> i64 {
|
||||
let is_modified_git_status = git_status.is_some_and(is_modified_status);
|
||||
if !is_modified_git_status {
|
||||
return 0;
|
||||
}
|
||||
|
||||
let now = self.get_now();
|
||||
let duration_since = now.saturating_sub(modified_time);
|
||||
|
||||
for i in 0..MODIFICATION_THRESHOLDS.len() {
|
||||
let (current_points, current_threshold) = MODIFICATION_THRESHOLDS[i];
|
||||
|
||||
if duration_since <= current_threshold {
|
||||
if i == 0 || duration_since == current_threshold {
|
||||
return current_points;
|
||||
}
|
||||
|
||||
let (prev_points, prev_threshold) = MODIFICATION_THRESHOLDS[i - 1];
|
||||
|
||||
let time_range = current_threshold - prev_threshold;
|
||||
let time_offset = duration_since - prev_threshold;
|
||||
let points_diff = prev_points - current_points;
|
||||
|
||||
let interpolated_score =
|
||||
prev_points - (points_diff * time_offset as i64) / time_range as i64;
|
||||
|
||||
return interpolated_score;
|
||||
}
|
||||
}
|
||||
|
||||
0
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
fn calculate_test_frecency_score(access_timestamps: &[u64], current_time: u64) -> i64 {
|
||||
let mut total_frecency = 0.0;
|
||||
|
||||
for &access_time in access_timestamps {
|
||||
let days_ago = (current_time.saturating_sub(access_time) as f64) / SECONDS_PER_DAY;
|
||||
let decay_factor = (-DECAY_CONSTANT * days_ago).exp();
|
||||
total_frecency += decay_factor;
|
||||
}
|
||||
|
||||
let normalized_frecency = if total_frecency <= 20.0 {
|
||||
total_frecency
|
||||
} else {
|
||||
20.0 + (total_frecency - 10.0).sqrt()
|
||||
};
|
||||
|
||||
normalized_frecency.round() as i64
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_frecency_calculation() {
|
||||
let current_time = 1000000000; // Base timestamp
|
||||
|
||||
let score = calculate_test_frecency_score(&[], current_time);
|
||||
assert_eq!(score, 0);
|
||||
|
||||
let accesses = [current_time]; // Accessed right now
|
||||
let score = calculate_test_frecency_score(&accesses, current_time);
|
||||
assert_eq!(score, 1); // 1.0 decay factor = 1
|
||||
|
||||
let ten_days_seconds = 10 * 86400; // 10 days in seconds
|
||||
let accesses = [current_time - ten_days_seconds];
|
||||
let score = calculate_test_frecency_score(&accesses, current_time);
|
||||
assert_eq!(score, 1); // ~0.5 decay factor rounds to 1
|
||||
|
||||
let accesses = [
|
||||
current_time, // Today
|
||||
current_time - 86400, // 1 day ago
|
||||
current_time - 172800, // 2 days ago
|
||||
];
|
||||
let score = calculate_test_frecency_score(&accesses, current_time);
|
||||
assert!(score > 2 && score < 4, "Score: {}", score); // About 3 accesses with decay
|
||||
|
||||
let thirty_days = 30 * 86400;
|
||||
let accesses = [current_time - thirty_days]; // 30 days ago
|
||||
let score = calculate_test_frecency_score(&accesses, current_time);
|
||||
assert!(
|
||||
score < 2,
|
||||
"Old access should have minimal score, got: {}",
|
||||
score
|
||||
);
|
||||
|
||||
let recent_frequent = [current_time, current_time - 86400, current_time - 172800];
|
||||
let old_single = [current_time - ten_days_seconds];
|
||||
|
||||
let recent_score = calculate_test_frecency_score(&recent_frequent, current_time);
|
||||
let old_score = calculate_test_frecency_score(&old_single, current_time);
|
||||
|
||||
assert!(
|
||||
recent_score > old_score,
|
||||
"Recent frequent access ({}) should score higher than old single access ({})",
|
||||
recent_score,
|
||||
old_score
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_modification_score_interpolation() {
|
||||
let temp_dir = std::env::temp_dir().join("fff_test_interpolation");
|
||||
let _ = std::fs::remove_dir_all(&temp_dir);
|
||||
let tracker = FrecencyTracker::new(temp_dir.to_str().unwrap(), true).unwrap();
|
||||
|
||||
let current_time = tracker.get_now();
|
||||
let git_status = Some(git2::Status::WT_MODIFIED);
|
||||
|
||||
// At 5 minutes: should interpolate between 16 and 8 points
|
||||
let five_minutes_ago = current_time - (5 * 60);
|
||||
let score = tracker.get_modification_score(five_minutes_ago, git_status);
|
||||
|
||||
// Expected: 16 - (8 * 3 / 13) = 16 - 1 = 15 points
|
||||
// (time_offset = 5-2 = 3, time_range = 15-2 = 13, points_diff = 16-8 = 8)
|
||||
assert_eq!(score, 15, "5 minutes should interpolate to 15 points");
|
||||
|
||||
let two_minutes_ago = current_time - (2 * 60);
|
||||
let score = tracker.get_modification_score(two_minutes_ago, git_status);
|
||||
assert_eq!(score, 16, "2 minutes should be exactly 16 points");
|
||||
|
||||
let fifteen_minutes_ago = current_time - (15 * 60);
|
||||
let score = tracker.get_modification_score(fifteen_minutes_ago, git_status);
|
||||
assert_eq!(score, 8, "15 minutes should be exactly 8 points");
|
||||
|
||||
// At 12 hours: should interpolate between 4 and 2 points
|
||||
let twelve_hours_ago = current_time - (12 * 60 * 60);
|
||||
let score = tracker.get_modification_score(twelve_hours_ago, git_status);
|
||||
// Expected: 4 - (2 * 11 / 23) = 4 - 0 = 4 points (integer division)
|
||||
// (time_offset = 12-1 = 11 hours, time_range = 24-1 = 23 hours, points_diff = 4-2 = 2)
|
||||
assert_eq!(score, 4, "12 hours should interpolate to 4 points");
|
||||
|
||||
// at 18 hours for more significant interpolation
|
||||
let eighteen_hours_ago = current_time - (18 * 60 * 60);
|
||||
let score = tracker.get_modification_score(eighteen_hours_ago, git_status);
|
||||
// Expected: 4 - (2 * 17 / 23) = 4 - 1 = 3 points
|
||||
assert_eq!(score, 3, "18 hours should interpolate to 3 points");
|
||||
|
||||
let score = tracker.get_modification_score(five_minutes_ago, None);
|
||||
assert_eq!(score, 0, "No git status should return 0");
|
||||
|
||||
let _ = std::fs::remove_dir_all(&temp_dir);
|
||||
}
|
||||
}
|
||||
+143
-45
@@ -1,20 +1,40 @@
|
||||
use crate::error::Result;
|
||||
use ahash::AHashMap;
|
||||
use git2::{Repository, Status, StatusOptions};
|
||||
use std::{
|
||||
fmt::Debug,
|
||||
path::{Path, PathBuf},
|
||||
};
|
||||
use tracing::debug;
|
||||
|
||||
/// Represents a cache of a single git status query, if there is no
|
||||
/// status aka file is clear but it was specifically requested to updated
|
||||
/// the status is `None` otherwise contains only actual file statuses.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct GitStatusCache(Vec<(PathBuf, Status)>);
|
||||
pub(crate) fn default_status_options() -> StatusOptions {
|
||||
let mut opts = StatusOptions::new();
|
||||
opts.include_untracked(true)
|
||||
.recurse_untracked_dirs(true)
|
||||
.include_unmodified(true)
|
||||
.exclude_submodules(true);
|
||||
opts
|
||||
}
|
||||
|
||||
/// Status options for the initial scan / rescan.
|
||||
///
|
||||
/// Skips `include_unmodified` because every `FileItem` starts with
|
||||
/// `git_status: None` (== clean), so a missing cache entry already means
|
||||
/// "clean" — no need to ask libgit2 to enumerate every tracked path.
|
||||
/// Saves seconds on huge dirty trees (e.g. chromium with 400k+ entries).
|
||||
pub(crate) fn initial_scan_status_options() -> StatusOptions {
|
||||
let mut opts = StatusOptions::new();
|
||||
opts.include_untracked(true)
|
||||
.recurse_untracked_dirs(true)
|
||||
.exclude_submodules(true);
|
||||
opts
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Default)]
|
||||
pub(crate) struct GitStatusCache(AHashMap<PathBuf, Status>);
|
||||
|
||||
impl IntoIterator for GitStatusCache {
|
||||
type Item = (PathBuf, Status);
|
||||
type IntoIter = std::vec::IntoIter<Self::Item>;
|
||||
type IntoIter = <AHashMap<PathBuf, Status> as IntoIterator>::IntoIter;
|
||||
|
||||
fn into_iter(self) -> Self::IntoIter {
|
||||
self.0.into_iter()
|
||||
@@ -26,25 +46,27 @@ impl GitStatusCache {
|
||||
self.0.len()
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub fn lookup_status(&self, full_path: &Path) -> Option<Status> {
|
||||
self.0
|
||||
.binary_search_by(|(path, _)| path.as_path().cmp(full_path))
|
||||
.ok()
|
||||
.and_then(|idx| self.0.get(idx).map(|(_, status)| *status))
|
||||
self.0.get(full_path).copied()
|
||||
}
|
||||
|
||||
#[tracing::instrument(skip(repo, status_options))]
|
||||
fn read_status_impl(repo: &Repository, status_options: &mut StatusOptions) -> Result<Self> {
|
||||
let statuses = repo.statuses(Some(status_options))?;
|
||||
let Some(repo_path) = repo.workdir() else {
|
||||
return Ok(Self(vec![])); // repo is bare
|
||||
return Ok(Self(AHashMap::new())); // repo is bare
|
||||
};
|
||||
|
||||
let mut entries = Vec::with_capacity(statuses.len());
|
||||
let repo_path = crate::path_utils::normalize(repo_path.to_path_buf());
|
||||
|
||||
let mut entries = AHashMap::with_capacity(statuses.len());
|
||||
for entry in &statuses {
|
||||
if let Some(entry_path) = entry.path() {
|
||||
let full_path = repo_path.join(entry_path);
|
||||
entries.push((full_path, entry.status()));
|
||||
// libgit2 returns entry paths with forward slashes on every platform
|
||||
// fff stores native paths - meaning we have forward slash issue on windows
|
||||
let full_path = crate::path_utils::normalize(repo_path.join(entry_path));
|
||||
entries.insert(full_path, entry.status());
|
||||
}
|
||||
}
|
||||
|
||||
@@ -70,46 +92,38 @@ impl GitStatusCache {
|
||||
}
|
||||
}
|
||||
|
||||
#[tracing::instrument(skip(repo), level = tracing::Level::DEBUG)]
|
||||
#[tracing::instrument(skip(repo), fields(paths_count = paths.len()), level = tracing::Level::DEBUG)]
|
||||
pub fn git_status_for_paths<TPath: AsRef<Path> + Debug>(
|
||||
repo: &Repository,
|
||||
paths: &[TPath],
|
||||
) -> Result<Self> {
|
||||
if paths.is_empty() {
|
||||
return Ok(Self(vec![]));
|
||||
return Ok(Self(AHashMap::new()));
|
||||
}
|
||||
|
||||
let Some(workdir) = repo.workdir() else {
|
||||
return Ok(Self(vec![]));
|
||||
return Ok(Self(AHashMap::new()));
|
||||
};
|
||||
let workdir = crate::path_utils::normalize(workdir.to_path_buf());
|
||||
|
||||
// git pathspec is pretty slow and requires to walk the whole directory
|
||||
// so for a single file which is the most general use case we query directly the file
|
||||
if paths.len() == 1 {
|
||||
let full_path = paths[0].as_ref();
|
||||
let relative_path = full_path.strip_prefix(workdir)?;
|
||||
let relative_path = full_path.strip_prefix(&workdir)?;
|
||||
let status = repo.status_file(relative_path)?;
|
||||
|
||||
return Ok(Self(vec![(full_path.to_path_buf(), status)]));
|
||||
let mut map = AHashMap::with_capacity(1);
|
||||
map.insert(full_path.to_path_buf(), status);
|
||||
return Ok(Self(map));
|
||||
}
|
||||
|
||||
let mut status_options = StatusOptions::new();
|
||||
status_options
|
||||
.include_untracked(true)
|
||||
.recurse_untracked_dirs(true)
|
||||
// when reading partial status it's important to include all files requested
|
||||
.include_unmodified(true);
|
||||
|
||||
let mut status_options = default_status_options();
|
||||
for path in paths {
|
||||
status_options.pathspec(path.as_ref().strip_prefix(workdir)?);
|
||||
status_options.pathspec(path.as_ref().strip_prefix(&workdir)?);
|
||||
}
|
||||
|
||||
let git_status_cache = Self::read_status_impl(repo, &mut status_options)?;
|
||||
debug!(
|
||||
status_len = git_status_cache.statuses_len(),
|
||||
"Multiple files git status"
|
||||
);
|
||||
|
||||
Ok(git_status_cache)
|
||||
}
|
||||
}
|
||||
@@ -125,31 +139,115 @@ pub fn is_modified_status(status: Status) -> bool {
|
||||
)
|
||||
}
|
||||
|
||||
pub fn format_git_status(status: Option<Status>) -> &'static str {
|
||||
pub fn format_git_status_opt(status: Option<Status>) -> Option<&'static str> {
|
||||
match status {
|
||||
None => "clear",
|
||||
None => Some("clean"),
|
||||
Some(status) => {
|
||||
if status.contains(Status::WT_NEW) {
|
||||
"untracked"
|
||||
Some("untracked")
|
||||
} else if status.contains(Status::WT_MODIFIED) {
|
||||
"modified"
|
||||
Some("modified")
|
||||
} else if status.contains(Status::WT_DELETED) {
|
||||
"deleted"
|
||||
Some("deleted")
|
||||
} else if status.contains(Status::WT_RENAMED) {
|
||||
"renamed"
|
||||
Some("renamed")
|
||||
} else if status.contains(Status::INDEX_NEW) {
|
||||
"staged_new"
|
||||
Some("staged_new")
|
||||
} else if status.contains(Status::INDEX_MODIFIED) {
|
||||
"staged_modified"
|
||||
Some("staged_modified")
|
||||
} else if status.contains(Status::INDEX_DELETED) {
|
||||
"staged_deleted"
|
||||
Some("staged_deleted")
|
||||
} else if status.contains(Status::IGNORED) {
|
||||
"ignored"
|
||||
Some("ignored")
|
||||
} else if status.contains(Status::CURRENT) || status.is_empty() {
|
||||
"clean"
|
||||
Some("clean")
|
||||
} else {
|
||||
"unknown"
|
||||
None
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub fn format_git_status(status: Option<Status>) -> &'static str {
|
||||
format_git_status_opt(status).unwrap_or("unknown")
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use std::fs;
|
||||
use std::process::Command;
|
||||
use tempfile::TempDir;
|
||||
|
||||
fn git(dir: &Path, args: &[&str]) {
|
||||
let out = Command::new("git")
|
||||
.args(args)
|
||||
.current_dir(dir)
|
||||
.env("GIT_AUTHOR_NAME", "t")
|
||||
.env("GIT_AUTHOR_EMAIL", "t@t")
|
||||
.env("GIT_COMMITTER_NAME", "t")
|
||||
.env("GIT_COMMITTER_EMAIL", "t@t")
|
||||
.output()
|
||||
.unwrap();
|
||||
assert!(out.status.success(), "git {args:?} failed");
|
||||
}
|
||||
|
||||
/// Regression: on case-insensitive filesystems libgit2 returns
|
||||
/// statuses in a case-insensitive order. Our previous sorted-`Vec` +
|
||||
/// `binary_search_by(Path::cmp)` lookup silently missed entries
|
||||
/// because `Path::cmp` is byte-wise.
|
||||
///
|
||||
/// This test uses deliberately mixed-case filenames so the two
|
||||
/// orderings disagree, then checks every lookup succeeds.
|
||||
#[test]
|
||||
fn lookup_is_case_exact_regardless_of_libgit2_sort_order() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
// `std::fs::canonicalize` on Windows adds a `\\?\` UNC prefix that
|
||||
// libgit2's workdir string lacks. Use dunce so both sides match.
|
||||
let base = crate::path_utils::canonicalize(tmp.path()).unwrap();
|
||||
|
||||
// Mixed-case names that sort differently under byte-wise vs
|
||||
// case-insensitive comparators.
|
||||
let names = [
|
||||
"README.md",
|
||||
"a_lower.rs",
|
||||
"Z_upper.rs",
|
||||
"mixed_Case.txt",
|
||||
"nested/Inner_File.rs",
|
||||
];
|
||||
for n in &names {
|
||||
let p = base.join(n);
|
||||
fs::create_dir_all(p.parent().unwrap()).unwrap();
|
||||
fs::write(&p, format!("// {n}\n")).unwrap();
|
||||
}
|
||||
|
||||
git(&base, &["init", "-b", "main"]);
|
||||
git(&base, &["add", "-A"]);
|
||||
git(&base, &["commit", "-m", "seed", "--no-gpg-sign"]);
|
||||
|
||||
// Modify every file so they all end up in the status output as
|
||||
// WT_MODIFIED — guarantees a non-trivial map we have to look up.
|
||||
for n in &names {
|
||||
let p = base.join(n);
|
||||
fs::write(&p, format!("// {n}\n// edit\n")).unwrap();
|
||||
}
|
||||
|
||||
let repo = Repository::open(&base).unwrap();
|
||||
let paths: Vec<PathBuf> = names.iter().map(|n| base.join(n)).collect();
|
||||
let cache = GitStatusCache::git_status_for_paths(&repo, &paths).unwrap();
|
||||
|
||||
for (n, abs) in names.iter().zip(paths.iter()) {
|
||||
let status = cache.lookup_status(abs);
|
||||
assert!(
|
||||
status.is_some(),
|
||||
"lookup for {n} returned None; cache holds {} entries",
|
||||
cache.statuses_len(),
|
||||
);
|
||||
assert!(
|
||||
status.unwrap().contains(Status::WT_MODIFIED),
|
||||
"expected WT_MODIFIED for {n}, got {:?}",
|
||||
status
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+2100
-896
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,69 @@
|
||||
use std::path::Path;
|
||||
|
||||
pub(crate) const NON_GIT_IGNORED_DIRS: &[&str] = &[
|
||||
"node_modules",
|
||||
"__pycache__",
|
||||
"venv",
|
||||
".venv",
|
||||
// Rust (these are glob-only patterns for non_git_repo_overrides,
|
||||
// is_non_code_directory matches the "target" component separately)
|
||||
"target/debug",
|
||||
"target/release",
|
||||
"target/rust-analyzer",
|
||||
"target/criterion",
|
||||
];
|
||||
|
||||
#[cfg(target_os = "macos")]
|
||||
pub(crate) const PLATFORM_IGNORED_DIRS: &[&str] = &[
|
||||
"Library/Application Support",
|
||||
"Library/Caches",
|
||||
// App-group sandbox storage — used by iMessage, Photos, Notes, Calendar,
|
||||
// Electron apps, etc. for SQLite-WAL, LevelDB, protobuf files. These are
|
||||
// almost entirely extension-less binary files (~80k on a typical $HOME)
|
||||
// that never need to appear in a fuzzy or grep search.
|
||||
"Library/Group Containers",
|
||||
"Library/Containers",
|
||||
];
|
||||
|
||||
#[cfg(target_os = "windows")]
|
||||
pub(crate) const PLATFORM_IGNORED_DIRS: &[&str] = &[
|
||||
"bin/Debug",
|
||||
"bin/Release",
|
||||
"Program Files",
|
||||
"Program Files (x86)",
|
||||
"AppData/Local",
|
||||
"AppData/Roaming",
|
||||
];
|
||||
|
||||
#[cfg(not(any(target_os = "macos", target_os = "windows")))]
|
||||
pub(crate) const PLATFORM_IGNORED_DIRS: &[&str] = &[];
|
||||
|
||||
pub(crate) fn non_git_repo_overrides(base_path: &Path) -> Option<ignore::overrides::Override> {
|
||||
use ignore::overrides::OverrideBuilder;
|
||||
|
||||
let mut builder = OverrideBuilder::new(base_path);
|
||||
for dir in NON_GIT_IGNORED_DIRS.iter().chain(PLATFORM_IGNORED_DIRS) {
|
||||
let pattern = format!("!**/{dir}/");
|
||||
if let Err(e) = builder.add(&pattern) {
|
||||
tracing::warn!("failed to add ignore pattern {pattern}: {e}");
|
||||
}
|
||||
}
|
||||
|
||||
builder.build().ok()
|
||||
}
|
||||
|
||||
pub(crate) fn is_non_code_directory(path: &Path) -> bool {
|
||||
let path_str = path.as_os_str().to_str().unwrap_or("");
|
||||
NON_GIT_IGNORED_DIRS
|
||||
.iter()
|
||||
.chain(PLATFORM_IGNORED_DIRS)
|
||||
.any(|&dir| {
|
||||
#[cfg(target_os = "windows")]
|
||||
let dir = dir.replace('/', std::path::MAIN_SEPARATOR_STR);
|
||||
#[cfg(target_os = "windows")]
|
||||
return path_str.contains(dir.as_str());
|
||||
|
||||
#[cfg(not(target_os = "windows"))]
|
||||
path_str.contains(dir)
|
||||
})
|
||||
}
|
||||
+146
-34
@@ -1,43 +1,155 @@
|
||||
//! fff-core - High-performance file finder library
|
||||
//! # FFF Search — High-performance file finder core
|
||||
//!
|
||||
//! This crate provides the core file indexing and fuzzy search functionality.
|
||||
//! It maintains global state for the file picker, frecency tracker, and query tracker.
|
||||
//! This crate provides the core search engine for [FFF (Fast File Finder)](https://github.com/dmtrKovalenko/fff.nvim).
|
||||
//! It includes filesystem indexing with real-time watching, fuzzy matching powered
|
||||
//! by [frizbee](https://docs.rs/neo_frizbee), frecency scoring backed by LMDB,
|
||||
//! and multi-mode grep search.
|
||||
//!
|
||||
//! ## Architecture
|
||||
//!
|
||||
//! - [`file_picker::FilePicker`] — Main entry point. Indexes a directory tree in a
|
||||
//! background thread, maintains a sorted file list, watches the filesystem for
|
||||
//! changes, and performs fuzzy search with frecency-weighted scoring.
|
||||
//! - [`frecency::FrecencyTracker`] — LMDB-backed database that tracks file access
|
||||
//! and modification patterns for intelligent result ranking.
|
||||
//! - [`query_tracker::QueryTracker`] — Tracks search query history and provides
|
||||
//! "combo-boost" scoring for repeatedly matched files.
|
||||
//! - [`grep`] — Live grep search supporting regex, plain-text, and fuzzy modes
|
||||
//! with optional constraint filtering.
|
||||
//! - [`git`] — Git status caching and repository detection.
|
||||
//!
|
||||
//! ## Shared State
|
||||
//!
|
||||
//! [`SharedFilePicker`], [`SharedFrecency`], and [`SharedQueryTracker`] are
|
||||
//! newtype wrappers around `Arc<RwLock<Option<T>>>` for thread-safe shared
|
||||
//! access. They provide `read()` / `write()` methods with built-in error
|
||||
//! conversion and convenience helpers like `wait_for_scan()`.
|
||||
//!
|
||||
//! ## Quick Start
|
||||
//!
|
||||
//! ```
|
||||
//! use fff_search::file_picker::FilePicker;
|
||||
//! use fff_search::frecency::FrecencyTracker;
|
||||
//! use fff_search::query_tracker::QueryTracker;
|
||||
//! use fff_search::{
|
||||
//! FFFMode, FilePickerOptions, FuzzySearchOptions, PaginationArgs, QueryParser,
|
||||
//! SharedFrecency, SharedFilePicker, SharedQueryTracker,
|
||||
//! };
|
||||
//!
|
||||
//! let shared_picker = SharedFilePicker::default();
|
||||
//! let shared_frecency = SharedFrecency::default();
|
||||
//! let shared_query_tracker = SharedQueryTracker::default();
|
||||
//!
|
||||
//! let tmp = std::env::temp_dir().join("fff-doctest");
|
||||
//! std::fs::create_dir_all(&tmp).unwrap();
|
||||
//!
|
||||
//! // 1. Optionally initialize frecency and query tracker databases
|
||||
//! let frecency = FrecencyTracker::open(tmp.join("frecency"))?;
|
||||
//! shared_frecency.init(frecency)?;
|
||||
//!
|
||||
//! let query_tracker = QueryTracker::open(tmp.join("queries"))?;
|
||||
//! shared_query_tracker.init(query_tracker)?;
|
||||
//!
|
||||
//! // 2. Init the file picker (spawns background scan + watcher)
|
||||
//! FilePicker::new_with_shared_state(
|
||||
//! shared_picker.clone(),
|
||||
//! shared_frecency.clone(),
|
||||
//! FilePickerOptions {
|
||||
//! base_path: ".".into(),
|
||||
//! mode: FFFMode::Ai,
|
||||
//! ..Default::default()
|
||||
//! },
|
||||
//! )?;
|
||||
//!
|
||||
//! // 3. Wait for scan
|
||||
//! shared_picker.wait_for_scan(std::time::Duration::from_secs(10));
|
||||
//!
|
||||
//! // 4. Search: lock the picker and query tracker
|
||||
//! let picker_guard = shared_picker.read()?;
|
||||
//! let picker = picker_guard.as_ref().unwrap();
|
||||
//! let qt_guard = shared_query_tracker.read()?;
|
||||
//!
|
||||
//! // 5. Parse the query and perform fuzzy search
|
||||
//! let parser = QueryParser::default();
|
||||
//! let query = parser.parse("lib.rs");
|
||||
//!
|
||||
//! let results = picker.fuzzy_search(
|
||||
//! &query,
|
||||
//! qt_guard.as_ref(),
|
||||
//! FuzzySearchOptions {
|
||||
//! max_threads: 0,
|
||||
//! current_file: None,
|
||||
//! pagination: PaginationArgs { offset: 0, limit: 50 },
|
||||
//! ..Default::default()
|
||||
//! },
|
||||
//! );
|
||||
//!
|
||||
//! assert!(results.total_matched > 0);
|
||||
//! assert!(results.items.first().unwrap().relative_path(picker).ends_with("lib.rs"));
|
||||
//!
|
||||
//! let _ = std::fs::remove_dir_all(&tmp);
|
||||
//! # Ok::<(), Box<dyn std::error::Error>>(())
|
||||
//! ```
|
||||
|
||||
mod background_watcher;
|
||||
pub mod constraints;
|
||||
mod db_healthcheck;
|
||||
mod scan;
|
||||
// public only for benchmarks — the inverted index is still re-exported via
|
||||
// `pub use bigram_filter::*` below for external consumers.
|
||||
#[doc(hidden)]
|
||||
pub mod bigram_filter;
|
||||
pub mod bigram_query;
|
||||
mod constraints;
|
||||
mod error;
|
||||
pub mod file_picker;
|
||||
pub mod frecency;
|
||||
pub mod git;
|
||||
pub mod grep;
|
||||
pub mod path_utils;
|
||||
pub mod query_tracker;
|
||||
pub mod score;
|
||||
mod score;
|
||||
mod sort_buffer;
|
||||
pub(crate) mod stable_vec;
|
||||
// this is pub only for benchmarks
|
||||
pub mod case_insensitive_memmem;
|
||||
|
||||
pub(crate) mod simd_path;
|
||||
|
||||
/// Core file picker: filesystem indexing, background watching, and fuzzy search.
|
||||
///
|
||||
/// See [`FilePicker`](file_picker::FilePicker) for the main entry point.
|
||||
pub mod file_picker;
|
||||
|
||||
/// Database-backed persistence: frecency, query history, LMDB plumbing.
|
||||
pub mod dbs;
|
||||
pub use dbs::frecency;
|
||||
|
||||
/// Git status caching and repository detection utilities.
|
||||
pub mod git;
|
||||
|
||||
/// Live grep search with regex, plain-text, and fuzzy matching modes.
|
||||
///
|
||||
/// Supports constraint filtering (file extensions, path segments, globs)
|
||||
/// and parallel execution via rayon.
|
||||
pub mod grep;
|
||||
|
||||
/// Tracing/logging initialization and panic hook setup.
|
||||
pub mod log;
|
||||
|
||||
/// Path manipulation utilities: cross platform canonicalization, tilde expansion, and
|
||||
/// directory distance penalties for search scoring.
|
||||
pub mod path_utils;
|
||||
|
||||
pub use dbs::query_tracker;
|
||||
|
||||
/// Core data types shared across the crate.
|
||||
pub mod types;
|
||||
|
||||
use file_picker::FilePicker;
|
||||
use frecency::FrecencyTracker;
|
||||
use once_cell::sync::Lazy;
|
||||
use query_tracker::QueryTracker;
|
||||
use std::sync::RwLock;
|
||||
mod ignore;
|
||||
/// Thread-safe shared handles for [`FilePicker`], [`FrecencyTracker`],
|
||||
/// and [`QueryTracker`].
|
||||
pub mod shared;
|
||||
|
||||
// Global state - same pattern as fff-nvim
|
||||
pub static FRECENCY: Lazy<RwLock<Option<FrecencyTracker>>> = Lazy::new(|| RwLock::new(None));
|
||||
pub static FILE_PICKER: Lazy<RwLock<Option<FilePicker>>> = Lazy::new(|| RwLock::new(None));
|
||||
pub static QUERY_TRACKER: Lazy<RwLock<Option<QueryTracker>>> = Lazy::new(|| RwLock::new(None));
|
||||
|
||||
// Re-export main types for convenience
|
||||
pub use db_healthcheck::{DbHealth, DbHealthChecker};
|
||||
pub use bigram_filter::*;
|
||||
pub use dbs::db_healthcheck::{DbHealth, DbHealthChecker};
|
||||
pub use error::{Error, Result};
|
||||
pub use file_picker::{FuzzySearchOptions, ScanProgress};
|
||||
pub use types::{FileItem, PaginationArgs, Score, ScoringContext, SearchResult};
|
||||
|
||||
// Re-export grep types
|
||||
pub use grep::{GrepMatch, GrepMode, GrepResult, GrepSearchOptions};
|
||||
|
||||
// Re-export query parser types (including Location which moved there)
|
||||
pub use fff_query_parser::{
|
||||
Constraint, FFFQuery, FuzzyQuery, Location, QueryParser, location::parse_location,
|
||||
};
|
||||
pub use fff_query_parser::*;
|
||||
pub use file_picker::*;
|
||||
pub use frecency::*;
|
||||
pub use grep::*;
|
||||
pub use query_tracker::*;
|
||||
pub use shared::*;
|
||||
pub use types::*;
|
||||
|
||||
@@ -0,0 +1,150 @@
|
||||
//! Shared logging utilities for FFF crates.
|
||||
//!
|
||||
//! Provides file-based tracing initialization and crash handlers (panic hook
|
||||
//! + SIGSEGV signal handler) that write diagnostics to both stderr and the
|
||||
//! configured log file.
|
||||
|
||||
use std::io;
|
||||
use std::path::{Path, PathBuf};
|
||||
use tracing_appender::non_blocking;
|
||||
use tracing_subscriber::fmt::format::FmtSpan;
|
||||
use tracing_subscriber::{EnvFilter, fmt, prelude::*};
|
||||
|
||||
static TRACING_INITIALIZED: std::sync::OnceLock<tracing_appender::non_blocking::WorkerGuard> =
|
||||
std::sync::OnceLock::new();
|
||||
|
||||
static CRASH_HANDLERS_INSTALLED: std::sync::OnceLock<()> = std::sync::OnceLock::new();
|
||||
|
||||
/// The log file path set by `init_tracing`. Crash handlers append to this file.
|
||||
static LOG_FILE_PATH: std::sync::OnceLock<PathBuf> = std::sync::OnceLock::new();
|
||||
|
||||
fn write_crash_report(header: &str, body: &str) {
|
||||
let msg = format!(
|
||||
"\n=== CRASH {} ===\n{}\n=== CRASH END {} ===\n",
|
||||
header, body, header
|
||||
);
|
||||
|
||||
let _ = std::io::Write::write_all(&mut std::io::stderr(), msg.as_bytes());
|
||||
|
||||
if let Some(path) = LOG_FILE_PATH.get() {
|
||||
let _ = std::fs::OpenOptions::new()
|
||||
.create(true)
|
||||
.append(true)
|
||||
.open(path)
|
||||
.and_then(|mut f| std::io::Write::write_all(&mut f, msg.as_bytes()));
|
||||
}
|
||||
}
|
||||
|
||||
extern "C" fn sigsegv_handler(sig: libc::c_int) {
|
||||
let bt = std::backtrace::Backtrace::force_capture();
|
||||
write_crash_report("SIGSEGV", &format!("signal {}\n{}", sig, bt));
|
||||
|
||||
unsafe {
|
||||
libc::signal(sig, libc::SIG_DFL);
|
||||
libc::raise(sig);
|
||||
}
|
||||
}
|
||||
|
||||
/// Install both the panic hook and the SIGSEGV signal handler.
|
||||
pub fn install_panic_hook() {
|
||||
CRASH_HANDLERS_INSTALLED.get_or_init(|| {
|
||||
let default_panic = std::panic::take_hook();
|
||||
std::panic::set_hook(Box::new(move |panic_info| {
|
||||
let message = if let Some(s) = panic_info.payload().downcast_ref::<&str>() {
|
||||
s.to_string()
|
||||
} else if let Some(s) = panic_info.payload().downcast_ref::<String>() {
|
||||
s.clone()
|
||||
} else {
|
||||
"Unknown panic payload".to_string()
|
||||
};
|
||||
|
||||
let location = panic_info
|
||||
.location()
|
||||
.map(|l| format!("{}:{}:{}", l.file(), l.line(), l.column()))
|
||||
.unwrap_or_else(|| "unknown location".to_string());
|
||||
|
||||
tracing::error!(
|
||||
panic.message = %message,
|
||||
panic.location = %location,
|
||||
"PANIC occurred in FFF"
|
||||
);
|
||||
|
||||
write_crash_report(
|
||||
"RUST PANIC",
|
||||
&format!("Message: {}\nLocation: {}", message, location),
|
||||
);
|
||||
default_panic(panic_info);
|
||||
}));
|
||||
|
||||
unsafe {
|
||||
libc::signal(
|
||||
libc::SIGSEGV,
|
||||
sigsegv_handler as *const () as libc::sighandler_t,
|
||||
);
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
/// Parse a log level string into a `tracing::Level`.
|
||||
pub fn parse_log_level(level: Option<&str>) -> tracing::Level {
|
||||
match level.as_ref().map(|s| s.trim().to_lowercase()).as_deref() {
|
||||
Some("trace") => tracing::Level::TRACE,
|
||||
Some("debug") => tracing::Level::DEBUG,
|
||||
Some("info") => tracing::Level::INFO,
|
||||
Some("warn") => tracing::Level::WARN,
|
||||
Some("error") => tracing::Level::ERROR,
|
||||
_ => tracing::Level::INFO,
|
||||
}
|
||||
}
|
||||
|
||||
/// Initialize tracing with a single log file.
|
||||
pub fn init_tracing(log_file_path: &str, log_level: Option<&str>) -> Result<String, io::Error> {
|
||||
let log_path = Path::new(log_file_path);
|
||||
if let Some(parent) = log_path.parent() {
|
||||
std::fs::create_dir_all(parent)?;
|
||||
}
|
||||
|
||||
let _ = LOG_FILE_PATH.set(log_path.to_path_buf());
|
||||
install_panic_hook();
|
||||
|
||||
let file_appender = std::fs::OpenOptions::new()
|
||||
.create(true)
|
||||
.write(true)
|
||||
.truncate(true) // truncates a file on restart (instead of appending)
|
||||
.open(log_path)?;
|
||||
|
||||
let level = parse_log_level(log_level);
|
||||
|
||||
TRACING_INITIALIZED.get_or_init(|| {
|
||||
let (non_blocking_appender, guard) = non_blocking(file_appender);
|
||||
|
||||
let subscriber = tracing_subscriber::registry()
|
||||
.with(
|
||||
fmt::layer()
|
||||
.with_writer(non_blocking_appender)
|
||||
.with_target(true)
|
||||
.with_thread_ids(false)
|
||||
.with_thread_names(false)
|
||||
.with_ansi(false)
|
||||
.with_span_events(FmtSpan::NEW | FmtSpan::CLOSE),
|
||||
)
|
||||
.with(
|
||||
EnvFilter::builder()
|
||||
.with_default_directive(level.into())
|
||||
.from_env_lossy(),
|
||||
);
|
||||
|
||||
if let Err(e) = tracing::subscriber::set_global_default(subscriber) {
|
||||
eprintln!("Failed to set tracing subscriber: {}", e);
|
||||
} else {
|
||||
tracing::info!(
|
||||
"FFF tracing initialized with log file: {}",
|
||||
log_path.display()
|
||||
);
|
||||
}
|
||||
|
||||
guard
|
||||
});
|
||||
|
||||
Ok(log_file_path.to_string())
|
||||
}
|
||||
@@ -1,12 +1,5 @@
|
||||
//! Path utility functions for file picker scoring
|
||||
|
||||
use std::path::{Path, PathBuf};
|
||||
|
||||
/// Canonicalize a path, resolving symlinks and producing an absolute path.
|
||||
///
|
||||
/// On Windows, uses `dunce::canonicalize` to avoid the `\\?\` extended-length path prefix
|
||||
/// that `std::fs::canonicalize` produces. Neovim cannot open paths with this prefix.
|
||||
/// On other platforms, delegates directly to `std::fs::canonicalize`.
|
||||
#[cfg(windows)]
|
||||
pub fn canonicalize(path: impl AsRef<Path>) -> std::io::Result<PathBuf> {
|
||||
dunce::canonicalize(path)
|
||||
@@ -17,53 +10,88 @@ pub fn canonicalize(path: impl AsRef<Path>) -> std::io::Result<PathBuf> {
|
||||
std::fs::canonicalize(path)
|
||||
}
|
||||
|
||||
/// Calculate distance penalty based on directory proximity
|
||||
/// Returns a negative penalty score based on how far the candidate is from the current file
|
||||
pub fn calculate_distance_penalty(current_file: Option<&str>, candidate_path: &str) -> i32 {
|
||||
let Some(ref current_path) = current_file else {
|
||||
return 0; // No penalty if no current file
|
||||
};
|
||||
/// Git requires a normalized forward-slashed paths on windows
|
||||
#[cfg(windows)]
|
||||
pub fn normalize(path: PathBuf) -> PathBuf {
|
||||
let as_str = path.to_string_lossy();
|
||||
let with_backslashes: String = as_str.replace('/', "\\");
|
||||
let buf = PathBuf::from(with_backslashes);
|
||||
dunce::canonicalize(&buf).unwrap_or(buf)
|
||||
}
|
||||
|
||||
let current_dir = if let Some(parent) = std::path::Path::new(current_path).parent() {
|
||||
parent.to_string_lossy().to_string()
|
||||
} else {
|
||||
String::new()
|
||||
};
|
||||
#[cfg(not(windows))]
|
||||
pub fn normalize(path: PathBuf) -> PathBuf {
|
||||
path
|
||||
}
|
||||
|
||||
let candidate_dir = if let Some(parent) = std::path::Path::new(candidate_path).parent() {
|
||||
parent.to_string_lossy().to_string()
|
||||
} else {
|
||||
String::new()
|
||||
};
|
||||
#[cfg(windows)]
|
||||
pub fn expand_tilde(path: &str) -> PathBuf {
|
||||
return PathBuf::from(path);
|
||||
}
|
||||
|
||||
if current_dir == candidate_dir {
|
||||
return 0; // Same directory, no penalty
|
||||
#[cfg(not(windows))]
|
||||
pub fn expand_tilde(path: &str) -> PathBuf {
|
||||
if let Some(stripped) = path.strip_prefix("~/")
|
||||
&& let Some(home_dir) = dirs::home_dir()
|
||||
{
|
||||
return home_dir.join(stripped);
|
||||
}
|
||||
|
||||
let current_parts: Vec<&str> = current_dir
|
||||
.split(std::path::MAIN_SEPARATOR)
|
||||
.filter(|s| !s.is_empty())
|
||||
.collect();
|
||||
let candidate_parts: Vec<&str> = candidate_dir
|
||||
.split(std::path::MAIN_SEPARATOR)
|
||||
.filter(|s| !s.is_empty())
|
||||
.collect();
|
||||
PathBuf::from(path)
|
||||
}
|
||||
|
||||
let common_len = current_parts
|
||||
.iter()
|
||||
.zip(candidate_parts.iter())
|
||||
.take_while(|(a, b)| a == b)
|
||||
.count();
|
||||
/// Calculate distance penalty based on directory proximity.
|
||||
/// Returns a negative penalty score based on how far the candidate is from the current file.
|
||||
///
|
||||
/// `candidate_dir` is the directory portion of the candidate path (e.g. `"src/components/"`).
|
||||
/// It may have a trailing `/` which is stripped internally.
|
||||
///
|
||||
/// Zero-allocation: walks both directory part iterators in lockstep.
|
||||
pub fn calculate_distance_penalty(current_file: Option<&str>, candidate_dir: &str) -> i32 {
|
||||
let Some(current_path) = current_file else {
|
||||
return 0;
|
||||
};
|
||||
|
||||
let current_depth_from_common = current_parts.len() - common_len;
|
||||
let current_dir = Path::new(current_path).parent().unwrap_or(Path::new(""));
|
||||
let candidate = Path::new(candidate_dir);
|
||||
|
||||
if current_depth_from_common == 0 {
|
||||
return 0; // Current file is at the common ancestor level
|
||||
if current_dir == candidate {
|
||||
return 0;
|
||||
}
|
||||
|
||||
let penalty = -(current_depth_from_common as i32);
|
||||
let mut current_parts = current_dir.components();
|
||||
let mut candidate_parts = candidate.components();
|
||||
|
||||
penalty.max(-20)
|
||||
let mut common_len = 0usize;
|
||||
let mut current_total = 0usize;
|
||||
|
||||
loop {
|
||||
match (current_parts.next(), candidate_parts.next()) {
|
||||
(Some(a), Some(b)) => {
|
||||
current_total += 1;
|
||||
if a == b {
|
||||
common_len += 1;
|
||||
} else {
|
||||
current_total += current_parts.count();
|
||||
break;
|
||||
}
|
||||
}
|
||||
(Some(_), None) => {
|
||||
current_total += 1 + current_parts.count();
|
||||
break;
|
||||
}
|
||||
(None, _) => {
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
let depth_from_common = current_total - common_len;
|
||||
if depth_from_common == 0 {
|
||||
return 0;
|
||||
}
|
||||
|
||||
(-(depth_from_common as i32)).max(-20)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
@@ -73,16 +101,11 @@ mod tests {
|
||||
#[test]
|
||||
#[cfg(not(target_family = "windows"))]
|
||||
fn test_calculate_distance_penalty() {
|
||||
assert_eq!(
|
||||
calculate_distance_penalty(None, "examples/user/test/mod.rs"),
|
||||
0
|
||||
);
|
||||
// candidate_dir is now just the directory portion (with or without trailing /)
|
||||
assert_eq!(calculate_distance_penalty(None, "examples/user/test/"), 0);
|
||||
// Same directory
|
||||
assert_eq!(
|
||||
calculate_distance_penalty(
|
||||
Some("examples/user/test/main.rs"),
|
||||
"examples/user/test/mod.rs"
|
||||
),
|
||||
calculate_distance_penalty(Some("examples/user/test/main.rs"), "examples/user/test/"),
|
||||
0
|
||||
);
|
||||
//
|
||||
@@ -90,7 +113,7 @@ mod tests {
|
||||
assert_eq!(
|
||||
calculate_distance_penalty(
|
||||
Some("examples/user/test/subdir/file.rs"),
|
||||
"examples/user/test/mod.rs"
|
||||
"examples/user/test/"
|
||||
),
|
||||
-1
|
||||
);
|
||||
@@ -99,7 +122,7 @@ mod tests {
|
||||
assert_eq!(
|
||||
calculate_distance_penalty(
|
||||
Some("examples/user/test/dir1/file.rs"),
|
||||
"examples/user/test/dir2/mod.rs"
|
||||
"examples/user/test/dir2/"
|
||||
),
|
||||
-1
|
||||
);
|
||||
@@ -107,7 +130,7 @@ mod tests {
|
||||
assert_eq!(
|
||||
calculate_distance_penalty(
|
||||
Some("examples/audio-announce/src/lib/audio-announce.rs"),
|
||||
"examples/audio-announce/src/main.rs"
|
||||
"examples/audio-announce/src/"
|
||||
),
|
||||
-1
|
||||
);
|
||||
@@ -115,27 +138,27 @@ mod tests {
|
||||
assert_eq!(
|
||||
calculate_distance_penalty(
|
||||
Some("examples/audio-announce/src/audio-announce.rs"),
|
||||
"examples/pixel/src/main.rs"
|
||||
"examples/pixel/src/"
|
||||
),
|
||||
-2
|
||||
);
|
||||
|
||||
// Root level files
|
||||
assert_eq!(calculate_distance_penalty(Some("main.rs"), "lib.rs"), 0);
|
||||
// Root level files (empty dir)
|
||||
assert_eq!(calculate_distance_penalty(Some("main.rs"), ""), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
#[cfg(target_family = "windows")]
|
||||
fn distance_penalty_works_on_windows() {
|
||||
assert_eq!(
|
||||
calculate_distance_penalty(None, "examples\\user\\test\\mod.rs"),
|
||||
calculate_distance_penalty(None, "examples\\user\\test\\"),
|
||||
0
|
||||
);
|
||||
// Same directory
|
||||
assert_eq!(
|
||||
calculate_distance_penalty(
|
||||
Some("examples\\user\\test\\main.rs"),
|
||||
"examples\\user\\test\\mod.rs"
|
||||
"examples\\user\\test\\"
|
||||
),
|
||||
0
|
||||
);
|
||||
@@ -144,7 +167,7 @@ mod tests {
|
||||
assert_eq!(
|
||||
calculate_distance_penalty(
|
||||
Some("examples\\user\\test\\subdir\\file.rs"),
|
||||
"examples\\user\\test\\mod.rs"
|
||||
"examples\\user\\test\\"
|
||||
),
|
||||
-1
|
||||
);
|
||||
|
||||
@@ -0,0 +1,428 @@
|
||||
use std::path::PathBuf;
|
||||
use std::sync::Arc;
|
||||
use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering};
|
||||
|
||||
use rayon::prelude::*;
|
||||
use tracing::{error, info};
|
||||
|
||||
use crate::FileSync;
|
||||
use crate::background_watcher::BackgroundWatcher;
|
||||
use crate::bigram_filter::build_bigram_index;
|
||||
use crate::error::Error;
|
||||
use crate::file_picker::{BACKGROUND_THREAD_POOL, FFFMode};
|
||||
use crate::git::GitStatusCache;
|
||||
use crate::shared::{SharedFilePicker, SharedFrecency};
|
||||
use crate::simd_path::ArenaPtr;
|
||||
use crate::types::ContentCacheBudget;
|
||||
|
||||
#[derive(Clone, Default)]
|
||||
pub(crate) struct ScanSignals {
|
||||
/// Set to `true` while any scan phase is running
|
||||
pub(crate) scanning: Arc<AtomicBool>,
|
||||
/// Set to `true` once the filesystem watcher has been installed
|
||||
pub(crate) watcher_ready: Arc<AtomicBool>,
|
||||
/// Indicates that that owning picker was requested to shut down
|
||||
pub(crate) cancelled: Arc<AtomicBool>,
|
||||
/// Used to resolve conflicts if multiple rescans were triggered in a queue
|
||||
pub(crate) rescan_pending: Arc<AtomicBool>,
|
||||
/// Set by `post_scan_snapshot`, cleared by `PostScanSnapshot::drop`.
|
||||
/// DO NOT set or clear this manually — it is managed exclusively by the
|
||||
/// PostScanSnapshot lifecycle.
|
||||
pub(crate) post_scan_indexing_active: Arc<AtomicBool>,
|
||||
}
|
||||
|
||||
/// Which optional phases a scan should run.
|
||||
#[derive(Clone, Copy, Default, Debug)]
|
||||
pub(crate) struct ScanConfig {
|
||||
pub(crate) warmup: bool,
|
||||
pub(crate) content_indexing: bool,
|
||||
pub(crate) watch: bool,
|
||||
pub(crate) auto_cache_budget: bool,
|
||||
pub(crate) install_watcher: bool,
|
||||
pub(crate) follow_symlinks: bool,
|
||||
}
|
||||
|
||||
/// A fully-configured scan job ready to run on a background thread.
|
||||
///
|
||||
/// Build with [`ScanJob::from_picker`] (reads all state from the
|
||||
/// current `FilePicker`) or [`ScanJob::initial`] (for the bootstrap
|
||||
/// scan, before the picker is published to `SharedPicker`).
|
||||
pub(crate) struct ScanJob {
|
||||
shared_picker: SharedFilePicker,
|
||||
shared_frecency: SharedFrecency,
|
||||
base_path: PathBuf,
|
||||
mode: FFFMode,
|
||||
signals: ScanSignals,
|
||||
config: ScanConfig,
|
||||
/// Walker-maintained counter backing `get_scan_progress` on the UI
|
||||
/// side. Reset to 0 at scan start, incremented per-file by the
|
||||
/// walker. Shared `Arc` so the UI polls the same atomic.
|
||||
scanned_files_counter: Arc<AtomicUsize>,
|
||||
}
|
||||
|
||||
impl ScanJob {
|
||||
pub fn new_rescan(
|
||||
shared_picker: &SharedFilePicker,
|
||||
shared_frecency: &SharedFrecency,
|
||||
) -> Result<Option<Self>, Error> {
|
||||
let guard = shared_picker.read()?;
|
||||
let picker = guard.as_ref().ok_or(Error::FilePickerMissing)?;
|
||||
|
||||
if picker.is_scan_active()
|
||||
|| picker
|
||||
.signals
|
||||
.post_scan_indexing_active
|
||||
.load(Ordering::Acquire)
|
||||
{
|
||||
return Ok(None);
|
||||
}
|
||||
|
||||
let mode = picker.mode();
|
||||
let signals = picker.scan_signals();
|
||||
let scanned_files_counter = picker.scanned_files_counter();
|
||||
let base_path = picker.base_path().to_path_buf();
|
||||
|
||||
let new_scan_config = ScanConfig {
|
||||
warmup: picker.has_mmap_cache(),
|
||||
content_indexing: picker.has_content_indexing(),
|
||||
watch: picker.has_watcher(),
|
||||
auto_cache_budget: !picker.has_explicit_cache_budget(),
|
||||
install_watcher: false, // the watcher is independent of rescan, it is not restarting EVER
|
||||
follow_symlinks: picker.follows_symlinks(),
|
||||
};
|
||||
|
||||
drop(guard); // just a sanity check
|
||||
|
||||
Ok(Some(Self {
|
||||
mode,
|
||||
signals,
|
||||
base_path,
|
||||
scanned_files_counter,
|
||||
config: new_scan_config,
|
||||
shared_picker: shared_picker.clone(),
|
||||
shared_frecency: shared_frecency.clone(),
|
||||
}))
|
||||
}
|
||||
|
||||
pub fn new_initial(
|
||||
shared_picker: SharedFilePicker,
|
||||
shared_frecency: SharedFrecency,
|
||||
base_path: PathBuf,
|
||||
mode: FFFMode,
|
||||
signals: ScanSignals,
|
||||
scanned_files_counter: Arc<AtomicUsize>,
|
||||
config: ScanConfig,
|
||||
) -> Self {
|
||||
Self {
|
||||
shared_picker,
|
||||
shared_frecency,
|
||||
base_path,
|
||||
mode,
|
||||
signals,
|
||||
scanned_files_counter,
|
||||
config,
|
||||
}
|
||||
}
|
||||
|
||||
/// Spawn the job on a dedicated OS thread. Returns immediately.
|
||||
pub fn spawn(self) -> std::thread::JoinHandle<()> {
|
||||
self.signals.scanning.store(true, Ordering::Release);
|
||||
std::thread::Builder::new()
|
||||
.name("fff-scan".into())
|
||||
.spawn(move || self.run())
|
||||
.expect("failed to spawn fff-scan thread")
|
||||
}
|
||||
|
||||
fn run(self) {
|
||||
let Self {
|
||||
shared_picker,
|
||||
shared_frecency,
|
||||
base_path,
|
||||
mode,
|
||||
signals,
|
||||
scanned_files_counter,
|
||||
config,
|
||||
} = self;
|
||||
|
||||
let _scanning = ScanningGuard::new(&signals, config.install_watcher);
|
||||
|
||||
// Reset the UI-visible counter; the walker bumps it per file
|
||||
// and `get_scan_progress` reads it without locks.
|
||||
scanned_files_counter.store(0, Ordering::Relaxed);
|
||||
|
||||
// 1. Start git discovery and walk filesystem off-lock.
|
||||
let git_workdir = FileSync::discover_git_workdir(&base_path);
|
||||
let status_handle = git_workdir.clone().map(FileSync::spawn_git_status);
|
||||
let sync = match FileSync::walk_filesystem(
|
||||
&base_path,
|
||||
git_workdir.clone(),
|
||||
&scanned_files_counter,
|
||||
&shared_frecency,
|
||||
mode,
|
||||
config.follow_symlinks,
|
||||
) {
|
||||
Ok(sync) => sync,
|
||||
Err(e) => {
|
||||
error!(?e, "scan walk failed");
|
||||
return;
|
||||
}
|
||||
};
|
||||
|
||||
// 2. Brief write to install the freshly-walked file list.
|
||||
if let Ok(mut guard) = shared_picker.write()
|
||||
&& let Some(picker) = guard.as_mut()
|
||||
{
|
||||
if signals.cancelled.load(Ordering::Acquire) {
|
||||
info!("scan cancelled between walk and commit, discarding");
|
||||
return;
|
||||
}
|
||||
|
||||
let live_count = sync.live_count;
|
||||
picker.commit_new_sync(sync);
|
||||
|
||||
if config.auto_cache_budget && !picker.has_explicit_cache_budget() {
|
||||
picker.set_cache_budget(ContentCacheBudget::new_for_repo(live_count));
|
||||
}
|
||||
} else {
|
||||
error!("failed to install scan results into picker");
|
||||
return;
|
||||
}
|
||||
|
||||
// Files are now searchable — flip the scan signal *early* so
|
||||
// UI progress polls see the picker as "ready" while we run the
|
||||
// optional post-scan steps in the background.
|
||||
signals.scanning.store(false, Ordering::Relaxed);
|
||||
|
||||
// in case we do a rescan, we have to resubscribe a watcher to the new set of directories
|
||||
// all the already watched directories are not going to be resubscribed
|
||||
if !config.install_watcher && !signals.cancelled.load(Ordering::Acquire) {
|
||||
rescubscribe_watcher_post_scan(&shared_picker);
|
||||
}
|
||||
|
||||
let mut snapshot = if !signals.cancelled.load(Ordering::Acquire) {
|
||||
shared_picker.read().ok().and_then(|guard| {
|
||||
guard
|
||||
.as_ref()
|
||||
.and_then(|picker| unsafe { picker.post_scan_snapshot() })
|
||||
})
|
||||
} else {
|
||||
None
|
||||
};
|
||||
|
||||
// 3. Post-scan warmup + bigram build — runs in parallel with the
|
||||
// git-status thread to overlap the two expensive phases.
|
||||
if (config.warmup || config.content_indexing)
|
||||
&& !signals.cancelled.load(Ordering::Acquire)
|
||||
&& let Some(snap) = snapshot.as_ref()
|
||||
{
|
||||
Self::run_post_scan(&shared_picker, &signals, &config, snap);
|
||||
}
|
||||
|
||||
// 4. Join and git status, this HAS to be done after the post scan
|
||||
if !signals.cancelled.load(Ordering::Acquire)
|
||||
&& let Some(status_handle) = status_handle
|
||||
&& let Some(snapshot) = snapshot.as_mut()
|
||||
// THIS DOES WAIT for potentially very long status query
|
||||
&& let Ok(Some(git_status)) = status_handle.join()
|
||||
{
|
||||
apply_git_status_and_frecency(git_status, &shared_frecency, mode, snapshot);
|
||||
}
|
||||
|
||||
drop(snapshot); // SNAPSHOT SHOULD NOT BE USED AFTER THIS POINT
|
||||
|
||||
// 5. Install filesystem watcher (initial scan only).
|
||||
if config.install_watcher && config.watch && !signals.cancelled.load(Ordering::Acquire) {
|
||||
let shared_picker: &SharedFilePicker = &shared_picker;
|
||||
let shared_frecency: &SharedFrecency = &shared_frecency;
|
||||
let base_path: &std::path::Path = &base_path;
|
||||
|
||||
match BackgroundWatcher::new(
|
||||
base_path.to_path_buf(),
|
||||
git_workdir,
|
||||
shared_picker.clone(),
|
||||
shared_frecency.clone(),
|
||||
mode,
|
||||
) {
|
||||
Ok(watcher) => {
|
||||
if let Ok(mut guard) = shared_picker.write()
|
||||
&& let Some(picker) = guard.as_mut()
|
||||
{
|
||||
picker.background_watcher = Some(watcher);
|
||||
}
|
||||
}
|
||||
Err(e) => error!(?e, "failed to initialize background watcher"),
|
||||
};
|
||||
}
|
||||
|
||||
// 6. Drain any rescan that arrived while we were busy.
|
||||
// if user initiated a new rescan we had no way to cancel current post scan, so do it again
|
||||
if !signals.cancelled.load(Ordering::Acquire)
|
||||
&& signals.rescan_pending.swap(false, Ordering::AcqRel)
|
||||
{
|
||||
match Self::new_rescan(&shared_picker, &shared_frecency) {
|
||||
Ok(Some(follow_up)) => {
|
||||
info!("Rescheduling deferred rescan after current scan finished");
|
||||
follow_up.spawn();
|
||||
}
|
||||
Ok(None) => {
|
||||
// this should be practically impossible because we do not have any
|
||||
// queue, but if somehow a new rescan was triggered JUST IN THIS MOMENT
|
||||
// just ignore it because the ongoing one is fresh enough
|
||||
tracing::warn!("Post scan was re-triggered, ignoring");
|
||||
}
|
||||
Err(e) => {
|
||||
error!(?e, "Failed to reschedule deferred rescan");
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// THIS IS VERY VERY IMPORTANT THAT ANYTHING INSIDE THIS FUNCTION TO NOT READ ANYTHING CLEARABLE OUTSIDE
|
||||
/// this is a very silly off lock implementation that actually matters, and that's why it is crafted
|
||||
/// to never read anything from the picker, it can only WRITE information using single instructions
|
||||
///
|
||||
/// Things that are safe and immutable - file list, indexes of files, paths, and signals.
|
||||
#[tracing::instrument(skip_all, fields(warmup = ?config.warmup, indexing = ?config.content_indexing))]
|
||||
fn run_post_scan(
|
||||
shared_picker: &SharedFilePicker,
|
||||
signals: &ScanSignals,
|
||||
config: &ScanConfig,
|
||||
unsafe_snapshot: &crate::file_picker::PostScanUnsafeSnapshot,
|
||||
) {
|
||||
let arena = unsafe_snapshot
|
||||
.arena
|
||||
.as_ref()
|
||||
.map(|s| s.as_arena_ptr())
|
||||
.unwrap_or(ArenaPtr::null());
|
||||
let _budget: &ContentCacheBudget = &unsafe_snapshot.budget;
|
||||
let files: &[crate::types::FileItem] = &unsafe_snapshot.files[..unsafe_snapshot.base_count];
|
||||
|
||||
if signals.cancelled.load(Ordering::Acquire) {
|
||||
return;
|
||||
}
|
||||
|
||||
if config.content_indexing {
|
||||
let indexable_files = &files[..unsafe_snapshot.indexable_count.min(files.len())];
|
||||
let index = build_bigram_index(indexable_files, &unsafe_snapshot.base_path, arena);
|
||||
|
||||
if let Ok(mut guard) = shared_picker.write()
|
||||
&& let Some(picker) = guard.as_mut()
|
||||
{
|
||||
picker.set_bigram_index(index);
|
||||
}
|
||||
}
|
||||
|
||||
// Skipped as potentially unsafe - figure this out later
|
||||
// if config.warmup && !signals.cancelled.load(Ordering::Acquire) {
|
||||
// warmup_mmaps(files, budget, &unsafe_snapshot.base_path, arena);
|
||||
// }
|
||||
}
|
||||
}
|
||||
|
||||
/// RAII helper that flips the `scanning` signal on construction and
|
||||
/// resets it on drop (so early-returns can't leave it stuck on `true`).
|
||||
/// Also drives the `watcher_ready` signal on the initial-scan path.
|
||||
struct ScanningGuard<'a> {
|
||||
signals: &'a ScanSignals,
|
||||
release_watcher_ready_on_drop: bool,
|
||||
}
|
||||
|
||||
impl<'a> ScanningGuard<'a> {
|
||||
fn new(signals: &'a ScanSignals, release_watcher_ready_on_drop: bool) -> Self {
|
||||
signals.scanning.store(true, Ordering::Relaxed);
|
||||
Self {
|
||||
signals,
|
||||
release_watcher_ready_on_drop,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl Drop for ScanningGuard<'_> {
|
||||
fn drop(&mut self) {
|
||||
self.signals.scanning.store(false, Ordering::Relaxed);
|
||||
if self.release_watcher_ready_on_drop {
|
||||
self.signals.watcher_ready.store(true, Ordering::Release);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// If the scan encounters new directories created we have to add them to the watch list
|
||||
/// this is fine because the watcher does deduplicate the entries and doesn't add a lot of
|
||||
/// garbage notify watchers / fs events streams
|
||||
#[tracing::instrument(skip_all)]
|
||||
fn rescubscribe_watcher_post_scan(shared_picker: &SharedFilePicker) {
|
||||
let Ok(guard) = shared_picker.read() else {
|
||||
return;
|
||||
};
|
||||
let Some(picker) = guard.as_ref() else {
|
||||
return;
|
||||
};
|
||||
let Some(watcher) = picker.background_watcher.as_ref() else {
|
||||
return;
|
||||
};
|
||||
|
||||
picker.for_each_dir(|dir: &std::path::Path| {
|
||||
watcher.request_watch_dir(dir.to_path_buf());
|
||||
std::ops::ControlFlow::Continue(())
|
||||
});
|
||||
}
|
||||
|
||||
#[tracing::instrument(
|
||||
level = "debug",
|
||||
skip_all,
|
||||
fields(file_count = tracing::field::Empty, dirty_count = tracing::field::Empty),
|
||||
)]
|
||||
fn apply_git_status_and_frecency(
|
||||
git_cache: GitStatusCache,
|
||||
shared_frecency: &SharedFrecency,
|
||||
mode: FFFMode,
|
||||
unsafe_snapshot: &mut crate::file_picker::PostScanUnsafeSnapshot,
|
||||
) {
|
||||
let frecency = shared_frecency.read().ok();
|
||||
let frecency_ref = frecency.as_ref().and_then(|f| f.as_ref());
|
||||
|
||||
let base_count = unsafe_snapshot.base_count;
|
||||
let files: &mut [crate::types::FileItem] = &mut unsafe_snapshot.files[..base_count];
|
||||
// Dir frecency goes through per-entry `AtomicI32`; a shared slice is
|
||||
// enough and avoids any `&mut` aliasing against the Arc-shared buffer.
|
||||
let dirs: &[crate::types::DirItem] = &unsafe_snapshot.dirs;
|
||||
let arena = unsafe_snapshot
|
||||
.arena
|
||||
.as_ref()
|
||||
.map(|s| s.as_arena_ptr())
|
||||
.unwrap_or(ArenaPtr::null());
|
||||
|
||||
// Reset dir frecency before recomputation.
|
||||
for dir in dirs.iter() {
|
||||
dir.reset_frecency();
|
||||
}
|
||||
|
||||
BACKGROUND_THREAD_POOL.install(|| {
|
||||
files.par_iter_mut().for_each(|file| {
|
||||
if unsafe_snapshot.cancelled.load(Ordering::Relaxed) {
|
||||
return;
|
||||
}
|
||||
|
||||
let mut buf = [0u8; crate::simd_path::PATH_BUF_SIZE];
|
||||
let absolute_path =
|
||||
file.write_absolute_path(arena, &unsafe_snapshot.base_path, &mut buf);
|
||||
|
||||
file.git_status = git_cache.lookup_status(absolute_path);
|
||||
if let Some(frecency) = frecency_ref {
|
||||
let _ =
|
||||
file.update_frecency_scores(frecency, arena, &unsafe_snapshot.base_path, mode);
|
||||
}
|
||||
|
||||
let score = file.access_frecency_score as i32;
|
||||
if score > 0 {
|
||||
let dir_idx = file.parent_dir_index as usize;
|
||||
if let Some(dir) = dirs.get(dir_idx) {
|
||||
dir.update_frecency_if_larger(score);
|
||||
}
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
let span = tracing::Span::current();
|
||||
span.record("dirty_count", git_cache.statuses_len());
|
||||
}
|
||||
+1190
-307
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,411 @@
|
||||
use std::path::{Path, PathBuf};
|
||||
use std::sync::{Arc, RwLock, RwLockReadGuard, RwLockWriteGuard, Weak};
|
||||
use std::time::{Duration, Instant};
|
||||
|
||||
use crate::dbs::lmdb::spawn_lmdb_gc;
|
||||
use crate::error::Error;
|
||||
use crate::file_picker::FilePicker;
|
||||
use crate::frecency::FrecencyTracker;
|
||||
use crate::git::GitStatusCache;
|
||||
use crate::query_tracker::QueryTracker;
|
||||
use crate::scan::ScanJob;
|
||||
|
||||
/// Poll `.git/index.lock` until it disappears (git write completed), giving up
|
||||
/// after [`GIT_LOCK_MAX_WAIT`]. Used by [`SharedPicker::refresh_git_status`]
|
||||
/// to avoid reading a half-updated index when the watcher fires mid-`git add`.
|
||||
///
|
||||
/// The wait is bounded and cheap: the lock file is typically cleared within
|
||||
/// a few milliseconds of the git command exiting.
|
||||
fn wait_for_git_index_lock_release(git_root: &Path) {
|
||||
const GIT_LOCK_POLL: Duration = Duration::from_millis(10);
|
||||
const GIT_LOCK_MAX_WAIT: Duration = Duration::from_millis(500);
|
||||
|
||||
let lock = git_root.join(".git").join("index.lock");
|
||||
// Fast path: no lock present.
|
||||
if !lock.exists() {
|
||||
return;
|
||||
}
|
||||
let deadline = Instant::now() + GIT_LOCK_MAX_WAIT;
|
||||
while lock.exists() && Instant::now() < deadline {
|
||||
std::thread::sleep(GIT_LOCK_POLL);
|
||||
}
|
||||
if lock.exists() {
|
||||
tracing::warn!(
|
||||
"Proceeding with git status refresh despite lingering \
|
||||
.git/index.lock at {} — will retry once it clears",
|
||||
lock.display()
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/// Thread-safe shared handle to the [`FilePicker`] instance.
|
||||
/// This accumulates only asynchronous non-blocking operations against the
|
||||
/// file picker: creating, triggering various rescans and so on.
|
||||
///
|
||||
/// For blocking access use internal picker via `.read()` or `.write()`
|
||||
///
|
||||
/// ```ignore
|
||||
/// let shared_picker = SharedFilePicker::default();
|
||||
///
|
||||
/// if let Some(picker) = shared_picker.read()?.as_ref() {
|
||||
/// let files = picker.fuzzy_search(&query, options);
|
||||
/// println!("Found {} files", files.len());
|
||||
/// } else {
|
||||
/// println!("Picker not initialized");
|
||||
/// }
|
||||
/// ```
|
||||
#[derive(Clone, Default)]
|
||||
pub struct SharedFilePicker(pub(crate) Arc<SharedPickerInner>);
|
||||
|
||||
pub struct SharedPickerInner {
|
||||
picker: parking_lot::RwLock<Option<FilePicker>>,
|
||||
}
|
||||
|
||||
impl Default for SharedPickerInner {
|
||||
fn default() -> Self {
|
||||
Self {
|
||||
picker: parking_lot::RwLock::new(None),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Non-owning handle to a [`SharedPicker`].
|
||||
#[derive(Clone)]
|
||||
pub(crate) struct WeakFilePicker(Weak<SharedPickerInner>);
|
||||
|
||||
impl WeakFilePicker {
|
||||
/// Try to promote the weak handle back to a strong [`SharedPicker`].
|
||||
///
|
||||
/// Returns `None` once every strong `SharedPicker` clone has been
|
||||
/// dropped. Callers should treat that as "the picker is being
|
||||
/// torn down" and exit their current iteration cleanly.
|
||||
pub(crate) fn upgrade(&self) -> Option<SharedFilePicker> {
|
||||
self.0.upgrade().map(SharedFilePicker)
|
||||
}
|
||||
}
|
||||
|
||||
impl std::fmt::Debug for SharedFilePicker {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
f.debug_tuple("SharedPicker").field(&"..").finish()
|
||||
}
|
||||
}
|
||||
|
||||
impl SharedFilePicker {
|
||||
pub fn read(&self) -> Result<parking_lot::RwLockReadGuard<'_, Option<FilePicker>>, Error> {
|
||||
Ok(self.0.picker.read())
|
||||
}
|
||||
|
||||
pub fn write(&self) -> Result<parking_lot::RwLockWriteGuard<'_, Option<FilePicker>>, Error> {
|
||||
Ok(self.0.picker.write())
|
||||
}
|
||||
|
||||
/// Produce a non-owning handle to the same inner picker.
|
||||
/// Use it if you don't need to block internal threads from dropping while owning this ref
|
||||
pub(crate) fn weaken(&self) -> WeakFilePicker {
|
||||
WeakFilePicker(Arc::downgrade(&self.0))
|
||||
}
|
||||
|
||||
/// Return `true` if this is an instance of the picker that requires a complicated post-scan
|
||||
/// indexing/cache warmup job. The indexing is not crazy but it takes time.
|
||||
pub fn need_complex_rebuild(&self) -> bool {
|
||||
let guard = self.0.picker.read();
|
||||
guard
|
||||
.as_ref()
|
||||
.is_some_and(|p| p.has_mmap_cache() || p.has_content_indexing())
|
||||
}
|
||||
|
||||
/// Block until the background filesystem scan finishes.
|
||||
/// Returns `true` if scan completed, `false` on timeout.
|
||||
pub fn wait_for_scan(&self, timeout: Duration) -> bool {
|
||||
let signal = {
|
||||
let guard = self.0.picker.read();
|
||||
match &*guard {
|
||||
Some(picker) => Arc::clone(&picker.signals.scanning),
|
||||
None => return true,
|
||||
}
|
||||
};
|
||||
|
||||
let start = std::time::Instant::now();
|
||||
while signal.load(std::sync::atomic::Ordering::Acquire) {
|
||||
if start.elapsed() >= timeout {
|
||||
return false;
|
||||
}
|
||||
std::thread::sleep(Duration::from_millis(10));
|
||||
}
|
||||
true
|
||||
}
|
||||
|
||||
/// Block until the background file watcher is ready.
|
||||
/// Returns `true` if watcher ready, `false` on timeout.
|
||||
pub fn wait_for_watcher(&self, timeout: Duration) -> bool {
|
||||
let watch_ready_signal = {
|
||||
let guard = self.0.picker.read();
|
||||
match &*guard {
|
||||
Some(picker) => Arc::clone(&picker.signals.watcher_ready),
|
||||
None => return true,
|
||||
}
|
||||
};
|
||||
|
||||
let start = std::time::Instant::now();
|
||||
while !watch_ready_signal.load(std::sync::atomic::Ordering::Acquire) {
|
||||
if start.elapsed() >= timeout {
|
||||
return false;
|
||||
}
|
||||
std::thread::sleep(Duration::from_millis(10));
|
||||
}
|
||||
true
|
||||
}
|
||||
|
||||
/// Blocks until both the filesystem walk and post-scan indexing are done.
|
||||
/// Returns true once scanning=false AND post_scan_indexing_active=false.
|
||||
pub fn wait_for_indexing_complete(&self, timeout: Duration) -> bool {
|
||||
let (scanning, post_scan_active) = {
|
||||
let guard = self.0.picker.read();
|
||||
match &*guard {
|
||||
Some(picker) => (
|
||||
Arc::clone(&picker.signals.scanning),
|
||||
Arc::clone(&picker.signals.post_scan_indexing_active),
|
||||
),
|
||||
None => return true,
|
||||
}
|
||||
};
|
||||
|
||||
let start = std::time::Instant::now();
|
||||
loop {
|
||||
if start.elapsed() >= timeout {
|
||||
return false;
|
||||
}
|
||||
let s = scanning.load(std::sync::atomic::Ordering::Acquire);
|
||||
let p = post_scan_active.load(std::sync::atomic::Ordering::Acquire);
|
||||
if !s && !p {
|
||||
return true;
|
||||
}
|
||||
std::thread::sleep(Duration::from_millis(10));
|
||||
}
|
||||
}
|
||||
|
||||
/// Trigger a full filesystem rescan without blocking the caller.
|
||||
/// Performs a safe async rescan. Guarantees only single active rescan per picker.
|
||||
/// If many rescans requested the last one guaranteed to be finished.
|
||||
pub fn trigger_full_rescan_async(&self, shared_frecency: &SharedFrecency) -> Result<(), Error> {
|
||||
match ScanJob::new_rescan(self, shared_frecency)? {
|
||||
Some(job) => {
|
||||
job.spawn();
|
||||
}
|
||||
None => {
|
||||
// we can not abort the ongoing sync, but if the events
|
||||
if let Ok(guard) = self.read()
|
||||
&& let Some(picker) = guard.as_ref()
|
||||
{
|
||||
picker
|
||||
.scan_signals()
|
||||
.rescan_pending
|
||||
.store(true, std::sync::atomic::Ordering::Release);
|
||||
tracing::info!(
|
||||
"Full rescan requested while another scan is active — \
|
||||
deferred via rescan_pending flag"
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Refresh git statuses for all indexed files.
|
||||
pub fn refresh_git_status(&self, shared_frecency: &SharedFrecency) -> Result<usize, Error> {
|
||||
use tracing::debug;
|
||||
|
||||
let git_status = {
|
||||
let guard = self.read()?;
|
||||
let Some(ref picker) = *guard else {
|
||||
return Err(Error::FilePickerMissing);
|
||||
};
|
||||
|
||||
let git_root = picker.git_root().map(|p| p.to_path_buf());
|
||||
drop(guard); // updating git status could take very long time, there is not risky as we
|
||||
// do not allow any mutations and deletions of files from the sync
|
||||
|
||||
debug!(?git_root, "Refreshing git status for picker");
|
||||
|
||||
if let Some(ref root) = git_root {
|
||||
wait_for_git_index_lock_release(root);
|
||||
}
|
||||
|
||||
GitStatusCache::read_git_status(
|
||||
git_root.as_deref(),
|
||||
&mut crate::git::default_status_options(),
|
||||
)
|
||||
};
|
||||
|
||||
let mut guard = self.write()?;
|
||||
let picker = guard.as_mut().ok_or(Error::FilePickerMissing)?;
|
||||
|
||||
let statuses_count = if let Some(git_status) = git_status {
|
||||
let count = git_status.statuses_len();
|
||||
picker.update_git_statuses(git_status, shared_frecency)?;
|
||||
count
|
||||
} else {
|
||||
0
|
||||
};
|
||||
|
||||
Ok(statuses_count)
|
||||
}
|
||||
}
|
||||
|
||||
/// Thread-safe shared handle to the [`FrecencyTracker`] instance.
|
||||
#[derive(Clone)]
|
||||
pub struct SharedFrecency {
|
||||
inner: Arc<RwLock<Option<FrecencyTracker>>>,
|
||||
enabled: bool,
|
||||
}
|
||||
|
||||
impl Default for SharedFrecency {
|
||||
fn default() -> Self {
|
||||
Self {
|
||||
inner: Arc::new(RwLock::new(None)),
|
||||
enabled: true,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl std::fmt::Debug for SharedFrecency {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
f.debug_tuple("SharedFrecency").field(&"..").finish()
|
||||
}
|
||||
}
|
||||
|
||||
impl SharedFrecency {
|
||||
/// Creates a disabled instance that silently ignores all writes.
|
||||
pub fn noop() -> Self {
|
||||
Self {
|
||||
inner: Arc::new(RwLock::new(None)),
|
||||
enabled: false,
|
||||
}
|
||||
}
|
||||
|
||||
pub fn read(&self) -> Result<RwLockReadGuard<'_, Option<FrecencyTracker>>, Error> {
|
||||
self.inner.read().map_err(|_| Error::AcquireFrecencyLock)
|
||||
}
|
||||
|
||||
pub fn write(&self) -> Result<RwLockWriteGuard<'_, Option<FrecencyTracker>>, Error> {
|
||||
self.inner.write().map_err(|_| Error::AcquireFrecencyLock)
|
||||
}
|
||||
|
||||
pub fn init(&self, tracker: FrecencyTracker) -> Result<(), Error> {
|
||||
if !self.enabled {
|
||||
return Ok(());
|
||||
}
|
||||
|
||||
{
|
||||
let mut guard = self.write()?;
|
||||
*guard = Some(tracker);
|
||||
}
|
||||
|
||||
// GC holds a read guard on this lock, so destroy / re-init wait
|
||||
// for it naturally — no join handle, no race against file removal.
|
||||
spawn_lmdb_gc(self.inner.clone());
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Drop the in-memory tracker and delete the on-disk database directory.
|
||||
///
|
||||
/// Acquires the write lock, ensuring all readers (including any active mmap
|
||||
/// access) are finished before the LMDB environment is closed and the files
|
||||
/// are removed.
|
||||
///
|
||||
/// Returns `Ok(Some(path))` with the deleted path, or `Ok(None)` if no
|
||||
/// tracker was initialized.
|
||||
pub fn destroy(&self) -> Result<Option<PathBuf>, Error> {
|
||||
let mut guard = self.write()?;
|
||||
let Some(tracker) = guard.take() else {
|
||||
return Ok(None);
|
||||
};
|
||||
let db_path = tracker.db_path().to_path_buf();
|
||||
// Drop closes the LMDB env and unmaps the files
|
||||
drop(tracker);
|
||||
drop(guard);
|
||||
std::fs::remove_dir_all(&db_path).map_err(|source| Error::RemoveDbDir {
|
||||
path: db_path.clone(),
|
||||
source,
|
||||
})?;
|
||||
Ok(Some(db_path))
|
||||
}
|
||||
}
|
||||
|
||||
/// Thread-safe shared handle to the [`QueryTracker`] instance.
|
||||
#[derive(Clone)]
|
||||
pub struct SharedQueryTracker {
|
||||
inner: Arc<RwLock<Option<QueryTracker>>>,
|
||||
enabled: bool,
|
||||
}
|
||||
|
||||
impl Default for SharedQueryTracker {
|
||||
fn default() -> Self {
|
||||
Self {
|
||||
inner: Arc::new(RwLock::new(None)),
|
||||
enabled: true,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl std::fmt::Debug for SharedQueryTracker {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
f.debug_tuple("SharedQueryTracker").field(&"..").finish()
|
||||
}
|
||||
}
|
||||
|
||||
impl SharedQueryTracker {
|
||||
/// Creates a disabled instance that silently ignores all writes.
|
||||
pub fn noop() -> Self {
|
||||
Self {
|
||||
inner: Arc::new(RwLock::new(None)),
|
||||
enabled: false,
|
||||
}
|
||||
}
|
||||
|
||||
pub fn read(&self) -> Result<RwLockReadGuard<'_, Option<QueryTracker>>, Error> {
|
||||
self.inner.read().map_err(|_| Error::AcquireFrecencyLock)
|
||||
}
|
||||
|
||||
pub fn write(&self) -> Result<RwLockWriteGuard<'_, Option<QueryTracker>>, Error> {
|
||||
self.inner.write().map_err(|_| Error::AcquireFrecencyLock)
|
||||
}
|
||||
|
||||
/// Initialize the query tracker + spawn GC in the background.
|
||||
/// No-op if this is a disabled instance.
|
||||
pub fn init(&self, tracker: QueryTracker) -> Result<(), Error> {
|
||||
if !self.enabled {
|
||||
return Ok(());
|
||||
}
|
||||
{
|
||||
let mut guard = self.write()?;
|
||||
*guard = Some(tracker);
|
||||
}
|
||||
|
||||
spawn_lmdb_gc(self.inner.clone());
|
||||
Ok(())
|
||||
}
|
||||
|
||||
///Drop the in-memory tracker and delete the on-disk database directory.
|
||||
///
|
||||
/// Acquires the write lock, ensuring all readers (including any active mmap
|
||||
/// access) are finished before the LMDB environment is closed and the files
|
||||
/// are removed.
|
||||
///
|
||||
/// Returns `Ok(Some(path))` with the deleted path, or `Ok(None)` if no
|
||||
/// tracker was initialized.
|
||||
pub fn destroy(&self) -> Result<Option<PathBuf>, Error> {
|
||||
let mut guard = self.write()?;
|
||||
let Some(tracker) = guard.take() else {
|
||||
return Ok(None);
|
||||
};
|
||||
let db_path = tracker.db_path().to_path_buf();
|
||||
drop(tracker);
|
||||
drop(guard);
|
||||
std::fs::remove_dir_all(&db_path).map_err(|source| Error::RemoveDbDir {
|
||||
path: db_path.clone(),
|
||||
source,
|
||||
})?;
|
||||
Ok(Some(db_path))
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,575 @@
|
||||
use ahash::AHashMap;
|
||||
use smallvec::SmallVec;
|
||||
use std::borrow::Cow;
|
||||
|
||||
/// SIMD chunk size in bytes (matches NEON/SSE2 register width).
|
||||
/// This must stay in sync with neo_frizbee's internal chunk size.
|
||||
pub(crate) const SIMD_CHUNK_BYTES: usize = 16;
|
||||
|
||||
/// 4 chunks = 64 bytes inline, covers ~85% of paths without heap fallback.
|
||||
const INLINE_CHUNKS: usize = 4;
|
||||
|
||||
pub(crate) type ChunkIndices = SmallVec<[u32; INLINE_CHUNKS]>;
|
||||
|
||||
#[derive(Clone, Copy)]
|
||||
pub struct ArenaPtr(pub(crate) *const u8);
|
||||
|
||||
// SAFETY: The arena is a read-only immutable part of file sync
|
||||
unsafe impl Send for ArenaPtr {}
|
||||
unsafe impl Sync for ArenaPtr {}
|
||||
|
||||
impl ArenaPtr {
|
||||
#[inline]
|
||||
pub fn new(ptr: *const u8) -> Self {
|
||||
Self(ptr)
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub fn null() -> Self {
|
||||
Self(std::ptr::null())
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub fn as_ptr(self) -> *const u8 {
|
||||
self.0
|
||||
}
|
||||
}
|
||||
|
||||
impl std::fmt::Debug for ArenaPtr {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
write!(f, "--arena-raw-pointer-0x({:?})", self.0)
|
||||
}
|
||||
}
|
||||
|
||||
#[repr(C, align(16))]
|
||||
#[derive(Clone, Copy)]
|
||||
pub(crate) struct SimdChunk(pub(crate) [u8; SIMD_CHUNK_BYTES]);
|
||||
|
||||
impl Default for SimdChunk {
|
||||
#[inline]
|
||||
fn default() -> Self {
|
||||
Self([0u8; SIMD_CHUNK_BYTES])
|
||||
}
|
||||
}
|
||||
|
||||
impl std::fmt::Debug for SimdChunk {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
// Show the actual bytes, trimming trailing zeros for readability
|
||||
let end = self.0.iter().rposition(|&b| b != 0).map_or(0, |i| i + 1);
|
||||
write!(f, "SimdChunk({:?})", &self.0[..end])
|
||||
}
|
||||
}
|
||||
|
||||
pub const PATH_BUF_SIZE: usize = 4096;
|
||||
|
||||
/// Indices into a shared `SimdChunk` arena representing a file path.
|
||||
///
|
||||
/// All read methods require an explicit `arena_base` pointer from the owning
|
||||
/// `ChunkedPathStore`. The struct itself contains no raw pointers to the arena
|
||||
#[derive(Clone)]
|
||||
pub(crate) struct ChunkedString {
|
||||
indices: ChunkIndices,
|
||||
pub byte_len: u16,
|
||||
/// Byte offset where the filename begins. 0 for root-level files.
|
||||
pub filename_offset: u16,
|
||||
}
|
||||
|
||||
impl ChunkedString {
|
||||
pub fn empty() -> Self {
|
||||
Self {
|
||||
indices: SmallVec::new(),
|
||||
byte_len: 0,
|
||||
filename_offset: 0,
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub fn new(indices: ChunkIndices, byte_len: u16, filename_offset: u16) -> Self {
|
||||
Self {
|
||||
indices,
|
||||
byte_len,
|
||||
filename_offset,
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
pub fn chunk_count(&self) -> usize {
|
||||
self.indices.len()
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub fn resolve_ptrs<'a>(
|
||||
&self,
|
||||
arena: ArenaPtr,
|
||||
buf: &'a mut [*const u8; 32],
|
||||
) -> &'a [*const u8] {
|
||||
let count = self.indices.len();
|
||||
let base = arena.as_ptr();
|
||||
for (i, &idx) in self.indices.iter().enumerate() {
|
||||
buf[i] = unsafe { base.add(idx as usize * SIMD_CHUNK_BYTES) };
|
||||
}
|
||||
&buf[..count]
|
||||
}
|
||||
|
||||
#[inline]
|
||||
fn write_slice_to_vec(
|
||||
indices: &[u32],
|
||||
base: *const u8,
|
||||
offset_in_chunk: usize,
|
||||
len: usize,
|
||||
vec: &mut Vec<u8>,
|
||||
) {
|
||||
let mut written = 0usize;
|
||||
for (i, &idx) in indices.iter().enumerate() {
|
||||
let src = unsafe { base.add(idx as usize * SIMD_CHUNK_BYTES) };
|
||||
let chunk_bytes = unsafe { core::slice::from_raw_parts(src, SIMD_CHUNK_BYTES) };
|
||||
let start = if i == 0 { offset_in_chunk } else { 0 };
|
||||
let end = SIMD_CHUNK_BYTES.min(start + (len - written));
|
||||
vec.extend_from_slice(&chunk_bytes[start..end]);
|
||||
written += end - start;
|
||||
}
|
||||
}
|
||||
|
||||
/// Return the filename portion as a `Cow<str>`.
|
||||
///
|
||||
/// When the filename starts at a chunk boundary and fits in one chunk we
|
||||
/// borrow directly from the arena (zero-copy). Otherwise we allocate.
|
||||
/// Filenames are almost always <=16 bytes so the fast path dominates.
|
||||
#[inline]
|
||||
pub fn filename_cow<'a>(&self, arena: ArenaPtr) -> Cow<'a, str> {
|
||||
let fname_offset = self.filename_offset as usize;
|
||||
let fname_len = self.byte_len as usize - fname_offset;
|
||||
if fname_len == 0 {
|
||||
return Cow::Borrowed("");
|
||||
}
|
||||
|
||||
let base = arena.as_ptr();
|
||||
let start_chunk = fname_offset / SIMD_CHUNK_BYTES;
|
||||
let offset_in_chunk = fname_offset % SIMD_CHUNK_BYTES;
|
||||
|
||||
if offset_in_chunk == 0 && fname_len <= SIMD_CHUNK_BYTES {
|
||||
let ptr = unsafe { base.add(self.indices[start_chunk] as usize * SIMD_CHUNK_BYTES) };
|
||||
let slice = unsafe { core::slice::from_raw_parts(ptr, fname_len) };
|
||||
return Cow::Borrowed(unsafe { core::str::from_utf8_unchecked(slice) });
|
||||
}
|
||||
|
||||
let mut out = String::with_capacity(fname_len);
|
||||
let needed_chunks = chunks_needed(offset_in_chunk + fname_len);
|
||||
Self::write_slice_to_vec(
|
||||
&self.indices[start_chunk..start_chunk + needed_chunks],
|
||||
base,
|
||||
offset_in_chunk,
|
||||
fname_len,
|
||||
unsafe { out.as_mut_vec() },
|
||||
);
|
||||
Cow::Owned(out)
|
||||
}
|
||||
|
||||
/// Truncates at `buf.len()` if exceeded -- use `[u8; PATH_BUF_SIZE]` to avoid.
|
||||
#[inline]
|
||||
pub fn read_to_buf<'a>(&self, arena: ArenaPtr, buf: &'a mut [u8]) -> &'a str {
|
||||
let total = (self.byte_len as usize).min(buf.len());
|
||||
let usable_chunks = total.div_ceil(SIMD_CHUNK_BYTES);
|
||||
let chunks_to_copy = usable_chunks.min(self.indices.len());
|
||||
let base = arena.as_ptr();
|
||||
|
||||
for (i, &idx) in self.indices[..chunks_to_copy].iter().enumerate() {
|
||||
let src = unsafe { base.add(idx as usize * SIMD_CHUNK_BYTES) };
|
||||
let dst_offset = i * SIMD_CHUNK_BYTES;
|
||||
let take = SIMD_CHUNK_BYTES.min(total - dst_offset);
|
||||
|
||||
unsafe {
|
||||
core::ptr::copy_nonoverlapping(src, buf.as_mut_ptr().add(dst_offset), take);
|
||||
}
|
||||
}
|
||||
|
||||
unsafe { core::str::from_utf8_unchecked(&buf[..total]) }
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub fn write_dir_to(&self, arena: ArenaPtr, out: &mut String) {
|
||||
out.clear();
|
||||
|
||||
let dir_len = self.filename_offset as usize;
|
||||
out.reserve(dir_len);
|
||||
let dir_chunks = chunks_needed(dir_len).min(self.indices.len());
|
||||
let base = arena.as_ptr();
|
||||
let vec = unsafe { out.as_mut_vec() };
|
||||
for (i, &idx) in self.indices[..dir_chunks].iter().enumerate() {
|
||||
let src = unsafe { base.add(idx as usize * SIMD_CHUNK_BYTES) };
|
||||
let take = SIMD_CHUNK_BYTES.min(dir_len - i * SIMD_CHUNK_BYTES);
|
||||
vec.extend_from_slice(unsafe { core::slice::from_raw_parts(src, take) });
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub fn write_filename_to(&self, arena: ArenaPtr, out: &mut String) {
|
||||
out.clear();
|
||||
|
||||
let fname_offset = self.filename_offset as usize;
|
||||
let fname_len = self.byte_len as usize - fname_offset;
|
||||
out.reserve(fname_len);
|
||||
let start_chunk = fname_offset / SIMD_CHUNK_BYTES;
|
||||
let offset_in_chunk = fname_offset % SIMD_CHUNK_BYTES;
|
||||
let needed_chunks = chunks_needed(offset_in_chunk + fname_len);
|
||||
Self::write_slice_to_vec(
|
||||
&self.indices[start_chunk..start_chunk + needed_chunks],
|
||||
arena.as_ptr(),
|
||||
offset_in_chunk,
|
||||
fname_len,
|
||||
unsafe { out.as_mut_vec() },
|
||||
);
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub fn write_to_string(&self, arena: ArenaPtr, out: &mut String) {
|
||||
out.clear();
|
||||
|
||||
let total = self.byte_len as usize;
|
||||
if total == 0 {
|
||||
return;
|
||||
}
|
||||
out.reserve(total);
|
||||
let base = arena.as_ptr();
|
||||
let vec = unsafe { out.as_mut_vec() };
|
||||
for (i, &idx) in self.indices.iter().enumerate() {
|
||||
let src = unsafe { base.add(idx as usize * SIMD_CHUNK_BYTES) };
|
||||
let take = SIMD_CHUNK_BYTES.min(total - i * SIMD_CHUNK_BYTES);
|
||||
vec.extend_from_slice(unsafe { core::slice::from_raw_parts(src, take) });
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl std::fmt::Debug for ChunkedString {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
f.debug_struct("ChunkedString")
|
||||
.field("indices", &self.indices.as_slice())
|
||||
.field("chunks", &self.indices.len())
|
||||
.field("byte_len", &self.byte_len)
|
||||
.field("filename_offset", &self.filename_offset)
|
||||
.finish()
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
const fn chunks_needed(byte_len: usize) -> usize {
|
||||
if byte_len == 0 {
|
||||
0
|
||||
} else {
|
||||
byte_len.div_ceil(SIMD_CHUNK_BYTES)
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug)]
|
||||
pub(crate) struct ChunkedPathStore {
|
||||
arena: Vec<SimdChunk>,
|
||||
}
|
||||
|
||||
// SAFETY: arena is immutable after construction. Pointers derived from it are
|
||||
// only read during scoring (no mutation, no reallocation).
|
||||
unsafe impl Send for ChunkedPathStore {}
|
||||
unsafe impl Sync for ChunkedPathStore {}
|
||||
|
||||
impl ChunkedPathStore {
|
||||
pub fn heap_bytes(&self) -> usize {
|
||||
self.arena.len() * SIMD_CHUNK_BYTES
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
fn unique_chunks(&self) -> usize {
|
||||
self.arena.len()
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub fn as_arena_ptr(&self) -> ArenaPtr {
|
||||
ArenaPtr::new(self.arena.as_ptr() as *const u8)
|
||||
}
|
||||
}
|
||||
|
||||
/// At runtime the builder should be split out from the store after `finish()`.
|
||||
#[derive(Clone, Debug)]
|
||||
pub(crate) struct ChunkedPathStoreBuilder {
|
||||
arena: Vec<SimdChunk>,
|
||||
chunk_dedup: AHashMap<[u8; SIMD_CHUNK_BYTES], u32>,
|
||||
}
|
||||
|
||||
impl ChunkedPathStoreBuilder {
|
||||
pub fn new(estimated_files: usize) -> Self {
|
||||
let est_chunks = estimated_files * 3;
|
||||
Self {
|
||||
arena: Vec::with_capacity(est_chunks / 2),
|
||||
chunk_dedup: AHashMap::with_capacity(est_chunks / 2),
|
||||
}
|
||||
}
|
||||
|
||||
pub fn finish(self) -> ChunkedPathStore {
|
||||
ChunkedPathStore { arena: self.arena }
|
||||
}
|
||||
|
||||
pub fn as_arena_ptr(&self) -> ArenaPtr {
|
||||
ArenaPtr::new(self.arena.as_ptr() as *const u8)
|
||||
}
|
||||
|
||||
/// Like [`add_file_immediate`] but for directory paths where the entire
|
||||
/// string is the "directory" portion (filename_offset == byte_len).
|
||||
pub fn add_dir_immediate(&mut self, dir_rel_path: &str) -> ChunkedString {
|
||||
self.add_file_immediate(dir_rel_path, dir_rel_path.len() as u16)
|
||||
}
|
||||
|
||||
pub fn add_file_immediate(&mut self, rel_path: &str, filename_offset: u16) -> ChunkedString {
|
||||
let path_bytes = rel_path.as_bytes();
|
||||
let byte_len = rel_path.len();
|
||||
let mut indices = ChunkIndices::with_capacity(chunks_needed(byte_len));
|
||||
|
||||
for chunk in path_bytes.chunks(SIMD_CHUNK_BYTES) {
|
||||
let mut chunk_bytes = [0u8; SIMD_CHUNK_BYTES];
|
||||
chunk_bytes[..chunk.len()].copy_from_slice(chunk);
|
||||
|
||||
let arena_idx = match self.chunk_dedup.get(&chunk_bytes) {
|
||||
Some(&idx) => idx,
|
||||
None => {
|
||||
let idx = self.arena.len() as u32;
|
||||
self.arena.push(SimdChunk(chunk_bytes));
|
||||
self.chunk_dedup.insert(chunk_bytes, idx);
|
||||
idx
|
||||
}
|
||||
};
|
||||
|
||||
indices.push(arena_idx);
|
||||
}
|
||||
|
||||
ChunkedString::new(indices, byte_len as u16, filename_offset)
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
pub(crate) fn build_chunked_path_store_from_strings(
|
||||
rel_paths: &[String],
|
||||
files: &[crate::types::FileItem],
|
||||
) -> (ChunkedPathStore, Vec<ChunkedString>) {
|
||||
assert_eq!(rel_paths.len(), files.len());
|
||||
let mut builder = ChunkedPathStoreBuilder::new(rel_paths.len());
|
||||
let strings: Vec<ChunkedString> = rel_paths
|
||||
.iter()
|
||||
.zip(files.iter())
|
||||
.map(|(rel_path, file)| builder.add_file_immediate(rel_path, file.path.filename_offset))
|
||||
.collect();
|
||||
(builder.finish(), strings)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
fn make_file_item(path: &str) -> crate::types::FileItem {
|
||||
let filename_start = path
|
||||
.rfind(std::path::is_separator)
|
||||
.map(|i| i + 1)
|
||||
.unwrap_or(0) as u16;
|
||||
crate::types::FileItem::new_raw(filename_start, 0, 0, None, false)
|
||||
}
|
||||
|
||||
fn build_test_store(
|
||||
paths: &[&str],
|
||||
) -> (
|
||||
ChunkedPathStore,
|
||||
Vec<ChunkedString>,
|
||||
Vec<crate::types::FileItem>,
|
||||
) {
|
||||
let mut files: Vec<crate::types::FileItem> =
|
||||
paths.iter().map(|p| make_file_item(p)).collect();
|
||||
let path_strings: Vec<String> = paths.iter().map(|p| p.to_string()).collect();
|
||||
let (store, strings) = build_chunked_path_store_from_strings(&path_strings, &files);
|
||||
for (i, file) in files.iter_mut().enumerate() {
|
||||
file.set_path(strings[i].clone());
|
||||
}
|
||||
(store, strings, files)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_chunked_store_empty() {
|
||||
let (store, strings, _files) = build_test_store(&[]);
|
||||
assert_eq!(strings.len(), 0);
|
||||
assert_eq!(store.unique_chunks(), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_chunked_store_basic() {
|
||||
let (store, strings, _files) =
|
||||
build_test_store(&["src/lib.rs", "src/main.rs", "Cargo.toml"]);
|
||||
let arena = store.as_arena_ptr();
|
||||
|
||||
assert_eq!(strings.len(), 3);
|
||||
assert!(store.unique_chunks() >= 2);
|
||||
|
||||
let mut buf = [0u8; 512];
|
||||
assert_eq!(
|
||||
strings[0].read_to_buf(arena, &mut buf).len(),
|
||||
"src/lib.rs".len()
|
||||
);
|
||||
assert_eq!(
|
||||
strings[2].read_to_buf(arena, &mut buf).len(),
|
||||
"Cargo.toml".len()
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_chunked_string_full_path() {
|
||||
let (store, strings, _files) = build_test_store(&["src/components/Button.tsx"]);
|
||||
let arena = store.as_arena_ptr();
|
||||
let cs = &strings[0];
|
||||
|
||||
let mut buf = [0u8; 512];
|
||||
assert_eq!(cs.read_to_buf(arena, &mut buf), "src/components/Button.tsx");
|
||||
assert_eq!(cs.byte_len, 25);
|
||||
assert_eq!(cs.filename_offset, 15);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_chunked_string_dir_and_filename() {
|
||||
let (store, strings, _files) = build_test_store(&["src/components/Button.tsx"]);
|
||||
let arena = store.as_arena_ptr();
|
||||
let cs = &strings[0];
|
||||
|
||||
let mut s = String::new();
|
||||
cs.write_dir_to(arena, &mut s);
|
||||
assert_eq!(s, "src/components/");
|
||||
cs.write_filename_to(arena, &mut s);
|
||||
assert_eq!(s, "Button.tsx");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_chunked_string_root_file() {
|
||||
let (store, strings, _files) = build_test_store(&["Cargo.toml"]);
|
||||
let arena = store.as_arena_ptr();
|
||||
let cs = &strings[0];
|
||||
|
||||
let mut s = String::new();
|
||||
cs.write_dir_to(arena, &mut s);
|
||||
assert_eq!(s, "");
|
||||
cs.write_filename_to(arena, &mut s);
|
||||
assert_eq!(s, "Cargo.toml");
|
||||
let mut buf = [0u8; 512];
|
||||
assert_eq!(cs.read_to_buf(arena, &mut buf), "Cargo.toml");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_chunked_string_resolve_ptrs() {
|
||||
let (store, strings, _files) = build_test_store(&["src/components/Button.tsx"]);
|
||||
let arena = store.as_arena_ptr();
|
||||
let cs = &strings[0];
|
||||
|
||||
let mut ptrs = [std::ptr::null::<u8>(); 32];
|
||||
let resolved = cs.resolve_ptrs(arena, &mut ptrs);
|
||||
assert_eq!(resolved.len(), 2); // 25 bytes = 2 chunks
|
||||
|
||||
// Verify we can read back the bytes
|
||||
let mut reconstructed = Vec::new();
|
||||
for (i, &ptr) in resolved.iter().enumerate() {
|
||||
let chunk = unsafe { std::slice::from_raw_parts(ptr, SIMD_CHUNK_BYTES) };
|
||||
let start = i * SIMD_CHUNK_BYTES;
|
||||
let take = SIMD_CHUNK_BYTES.min(25 - start);
|
||||
reconstructed.extend_from_slice(&chunk[..take]);
|
||||
}
|
||||
assert_eq!(
|
||||
std::str::from_utf8(&reconstructed).unwrap(),
|
||||
"src/components/Button.tsx"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_filename_cow_mid_chunk() {
|
||||
let (store, strings, _files) = build_test_store(&["src/components/Button.tsx"]);
|
||||
let arena = store.as_arena_ptr();
|
||||
let cs = &strings[0];
|
||||
|
||||
assert_eq!(cs.filename_offset, 15);
|
||||
assert_eq!(cs.byte_len, 25);
|
||||
|
||||
let fname = cs.filename_cow(arena);
|
||||
assert_eq!(&*fname, "Button.tsx");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_filename_cow_chunk_aligned() {
|
||||
let path = "0123456789abcdef/file.txt";
|
||||
let (store, strings, _files) = build_test_store(&[path]);
|
||||
let arena = store.as_arena_ptr();
|
||||
let cs = &strings[0];
|
||||
|
||||
assert_eq!(cs.filename_offset, 17);
|
||||
let fname = cs.filename_cow(arena);
|
||||
assert_eq!(&*fname, "file.txt");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_filename_cow_root_file() {
|
||||
let (store, strings, _files) = build_test_store(&["Cargo.toml"]);
|
||||
let arena = store.as_arena_ptr();
|
||||
let cs = &strings[0];
|
||||
|
||||
assert_eq!(cs.filename_offset, 0);
|
||||
let fname = cs.filename_cow(arena);
|
||||
assert_eq!(&*fname, "Cargo.toml");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_chunked_string_long_path() {
|
||||
let path = "very/deeply/nested/directory/structure/with/many/levels/file.txt";
|
||||
let (store, strings, _files) = build_test_store(&[path]);
|
||||
let arena = store.as_arena_ptr();
|
||||
let cs = &strings[0];
|
||||
|
||||
let mut buf = [0u8; 512];
|
||||
assert_eq!(cs.read_to_buf(arena, &mut buf), path);
|
||||
assert!(
|
||||
cs.chunk_count() <= 6,
|
||||
"should fit inline in ChunkIndices (INLINE_CHUNKS={})",
|
||||
INLINE_CHUNKS
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_chunked_string_clone() {
|
||||
let (store, strings, _files) = build_test_store(&["src/main.rs"]);
|
||||
let arena = store.as_arena_ptr();
|
||||
let cs = &strings[0];
|
||||
let cs2 = cs.clone();
|
||||
|
||||
let mut buf1 = [0u8; 512];
|
||||
let mut buf2 = [0u8; 512];
|
||||
assert_eq!(
|
||||
cs.read_to_buf(arena, &mut buf1),
|
||||
cs2.read_to_buf(arena, &mut buf2)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_chunked_string_full_path_roundtrip() {
|
||||
let paths = [
|
||||
"src/components/Button.tsx",
|
||||
"src/components/ui/DatePicker.tsx",
|
||||
"very/deeply/nested/directory/structure/file.txt",
|
||||
"Cargo.toml",
|
||||
"a.rs",
|
||||
];
|
||||
let (store, strings, _files) = build_test_store(&paths);
|
||||
let arena = store.as_arena_ptr();
|
||||
|
||||
for (i, expected) in paths.iter().enumerate() {
|
||||
let mut buf = [0u8; 512];
|
||||
let got = strings[i].read_to_buf(arena, &mut buf);
|
||||
assert_eq!(got, *expected, "full path roundtrip failed for file {i}");
|
||||
|
||||
let mut ds = String::new();
|
||||
let mut fs = String::new();
|
||||
strings[i].write_dir_to(arena, &mut ds);
|
||||
strings[i].write_filename_to(arena, &mut fs);
|
||||
assert_eq!(
|
||||
format!("{ds}{fs}"),
|
||||
*expected,
|
||||
"dir+fname mismatch for file {i}"
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,43 +1,56 @@
|
||||
//! Thread-local sort buffer management for glidesort optimization
|
||||
//!
|
||||
//! This module provides thread-local buffers for glidesort's with_buffer API,
|
||||
//! eliminating allocations in the hot path of fuzzy search operations.
|
||||
|
||||
use std::cell::RefCell;
|
||||
use parking_lot::Mutex;
|
||||
use std::mem::MaybeUninit;
|
||||
|
||||
// glidesort requires a buffer to allocate, we use one reused buffer as it can grow pretty big
|
||||
// for a large projects, this effectively saves 12kb of allocation on every search in linux repo
|
||||
thread_local! {
|
||||
static SORT_BUFFER: RefCell<Vec<u8>> = RefCell::new(Vec::with_capacity(1024));
|
||||
// this originally happen to be in TLS but there is a limit of TLS
|
||||
// + the storage itself is not free, so now we rely on the fact that most calls
|
||||
// are sequential in practice and allocate ONLY when we have a parallel access
|
||||
static SORT_BUFFER: Mutex<Vec<u8>> = Mutex::new(Vec::new());
|
||||
|
||||
fn ensure_capacity(buf: &mut Vec<u8>, required: usize) {
|
||||
if buf.capacity() < required {
|
||||
let len = buf.len();
|
||||
buf.reserve(required - len);
|
||||
}
|
||||
}
|
||||
|
||||
struct SharedSortBuf {
|
||||
guard: parking_lot::MutexGuard<'static, Vec<u8>>,
|
||||
}
|
||||
|
||||
impl SharedSortBuf {
|
||||
fn as_slice_mut<T>(&mut self, len: usize) -> &mut [MaybeUninit<T>] {
|
||||
let align = std::mem::align_of::<MaybeUninit<T>>();
|
||||
let size = std::mem::size_of::<MaybeUninit<T>>();
|
||||
let required = len.saturating_mul(size).saturating_add(align);
|
||||
ensure_capacity(&mut self.guard, required);
|
||||
|
||||
// SAFETY: the Vec<u8> is only 1-byte aligned, so we over-allocate by
|
||||
// `align` bytes and shift the pointer to satisfy T's alignment.
|
||||
// Callers never read uninitialised data through the returned slice.
|
||||
unsafe {
|
||||
let ptr = self.guard.as_mut_ptr();
|
||||
let offset = ptr.align_offset(align);
|
||||
debug_assert!(offset != usize::MAX && offset + len * size <= self.guard.capacity());
|
||||
std::slice::from_raw_parts_mut(ptr.add(offset) as *mut MaybeUninit<T>, len)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn try_lock_shared_buf() -> Option<SharedSortBuf> {
|
||||
SORT_BUFFER.try_lock().map(|guard| SharedSortBuf { guard })
|
||||
}
|
||||
|
||||
pub fn sort_with_buffer<T, F>(slice: &mut [T], compare: F)
|
||||
where
|
||||
F: FnMut(&T, &T) -> std::cmp::Ordering,
|
||||
{
|
||||
SORT_BUFFER.with(|buffer| {
|
||||
let mut buffer = buffer.borrow_mut();
|
||||
|
||||
// Calculate required buffer size in u8 units
|
||||
let size_of_t = std::mem::size_of::<MaybeUninit<T>>();
|
||||
let size_of_usize = std::mem::size_of::<u8>();
|
||||
let required_usizes = (slice.len() * size_of_t).div_ceil(size_of_usize);
|
||||
|
||||
// Ensure buffer has enough capacity
|
||||
if buffer.len() < required_usizes {
|
||||
buffer.resize(required_usizes, 0);
|
||||
match try_lock_shared_buf() {
|
||||
Some(mut buf) => {
|
||||
let typed = buf.as_slice_mut::<T>(slice.len());
|
||||
glidesort::sort_with_buffer_by(slice, typed, compare);
|
||||
}
|
||||
|
||||
// Cast u8 buffer to MaybeUninit<T> slice
|
||||
// SAFETY: u8 provides sufficient alignment for most types, and we've ensured
|
||||
// the buffer is large enough
|
||||
let typed_buffer = unsafe {
|
||||
std::slice::from_raw_parts_mut(buffer.as_mut_ptr() as *mut MaybeUninit<T>, slice.len())
|
||||
};
|
||||
|
||||
glidesort::sort_with_buffer_by(slice, typed_buffer, compare);
|
||||
});
|
||||
None => glidesort::sort_by(slice, compare),
|
||||
}
|
||||
}
|
||||
|
||||
pub fn sort_by_key_with_buffer<T, K, F>(slice: &mut [T], key_fn: F)
|
||||
@@ -45,28 +58,13 @@ where
|
||||
K: Ord,
|
||||
F: FnMut(&T) -> K,
|
||||
{
|
||||
SORT_BUFFER.with(|buffer| {
|
||||
let mut buffer = buffer.borrow_mut();
|
||||
|
||||
// Calculate required buffer size in u8 units
|
||||
let size_of_t = std::mem::size_of::<MaybeUninit<T>>();
|
||||
let size_of_usize = std::mem::size_of::<u8>();
|
||||
let required_usizes = (slice.len() * size_of_t).div_ceil(size_of_usize);
|
||||
|
||||
// Ensure buffer has enough capacity
|
||||
if buffer.len() < required_usizes {
|
||||
buffer.resize(required_usizes, 0);
|
||||
match try_lock_shared_buf() {
|
||||
Some(mut buf) => {
|
||||
let typed = buf.as_slice_mut::<T>(slice.len());
|
||||
glidesort::sort_with_buffer_by_key(slice, typed, key_fn);
|
||||
}
|
||||
|
||||
// Cast u8 buffer to MaybeUninit<T> slice
|
||||
// SAFETY: u8 provides sufficient alignment for most types, and we've ensured
|
||||
// the buffer is large enough
|
||||
let typed_buffer = unsafe {
|
||||
std::slice::from_raw_parts_mut(buffer.as_mut_ptr() as *mut MaybeUninit<T>, slice.len())
|
||||
};
|
||||
|
||||
glidesort::sort_with_buffer_by_key(slice, typed_buffer, key_fn);
|
||||
});
|
||||
None => glidesort::sort_by_key(slice, key_fn),
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
@@ -82,9 +80,9 @@ mod tests {
|
||||
|
||||
#[test]
|
||||
fn test_sort_by_key_with_buffer() {
|
||||
let mut data = vec![(2, "b"), (1, "a"), (3, "c")];
|
||||
sort_by_key_with_buffer(&mut data, |item| item.0);
|
||||
assert_eq!(data, vec![(1, "a"), (2, "b"), (3, "c")]);
|
||||
let mut data = vec![(1, 50), (2, 20), (3, 80), (4, 10), (5, 90)];
|
||||
sort_by_key_with_buffer(&mut data, |a| a.1);
|
||||
assert_eq!(data, vec![(4, 10), (2, 20), (1, 50), (3, 80), (5, 90)]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -94,19 +92,6 @@ mod tests {
|
||||
assert_eq!(data, vec![5, 4, 3, 2, 1]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_multiple_sorts_reuse_buffer() {
|
||||
// This test verifies that multiple sorts on the same thread reuse the buffer
|
||||
let mut data1 = vec![5, 2, 8, 1, 9];
|
||||
sort_with_buffer(&mut data1, |a, b| a.cmp(b));
|
||||
|
||||
let mut data2 = vec![15, 12, 18, 11, 19];
|
||||
sort_with_buffer(&mut data2, |a, b| a.cmp(b));
|
||||
|
||||
assert_eq!(data1, vec![1, 2, 5, 8, 9]);
|
||||
assert_eq!(data2, vec![11, 12, 15, 18, 19]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_empty_slice() {
|
||||
let mut data: Vec<i32> = vec![];
|
||||
@@ -121,13 +106,6 @@ mod tests {
|
||||
assert_eq!(data, vec![42]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_already_sorted() {
|
||||
let mut data = vec![1, 2, 3, 4, 5];
|
||||
sort_with_buffer(&mut data, |a, b| a.cmp(b));
|
||||
assert_eq!(data, vec![1, 2, 3, 4, 5]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_with_duplicates() {
|
||||
let mut data = vec![3, 1, 4, 1, 5, 9, 2, 6, 5];
|
||||
@@ -144,11 +122,10 @@ mod tests {
|
||||
|
||||
#[test]
|
||||
fn test_simple_descending() {
|
||||
// Simple test to verify highest scores come first
|
||||
let mut data = vec![100, 300, 200];
|
||||
sort_with_buffer(&mut data, |a, b| b.cmp(a));
|
||||
assert_eq!(data[0], 300, "Highest should be first");
|
||||
assert_eq!(data[1], 200, "Middle should be second");
|
||||
assert_eq!(data[2], 100, "Lowest should be last");
|
||||
assert_eq!(data[0], 300);
|
||||
assert_eq!(data[1], 200);
|
||||
assert_eq!(data[2], 100);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,177 @@
|
||||
use std::alloc::{self, Layout};
|
||||
use std::ptr::NonNull;
|
||||
use std::sync::Arc;
|
||||
use std::sync::atomic::{AtomicUsize, Ordering};
|
||||
|
||||
/// Vector that guarantees no re-alloc happening at runtime
|
||||
pub(crate) struct StableVec<T> {
|
||||
inner: Arc<StableBuf<T>>,
|
||||
}
|
||||
|
||||
struct StableBuf<T> {
|
||||
ptr: NonNull<T>,
|
||||
cap: usize,
|
||||
/// Atomic because:
|
||||
/// 1. `push(&self)` must mutate this through a shared `&StableBuf`,
|
||||
/// which requires interior mutability.
|
||||
/// 2. Arc clones (e.g. post-scan snapshots) read `len` outside the
|
||||
/// picker lock, concurrent with an appending writer. Acquire/Release
|
||||
/// on len is what makes "observed len ⇒ element bytes initialized"
|
||||
/// actually hold.
|
||||
///
|
||||
/// Arc wrapping only shares ownership of the buffer; it does NOT
|
||||
/// synchronize access to fields inside the shared buffer.
|
||||
len: AtomicUsize,
|
||||
}
|
||||
|
||||
// SAFETY: StableBuf is a thread-safe container when T is send + sync
|
||||
// There is another application level constraint: mutations are safe
|
||||
// when they are atomic updates, not read + update.
|
||||
unsafe impl<T: Send> Send for StableBuf<T> {}
|
||||
unsafe impl<T: Sync> Sync for StableBuf<T> {}
|
||||
|
||||
impl<T> Drop for StableBuf<T> {
|
||||
fn drop(&mut self) {
|
||||
let len = *self.len.get_mut();
|
||||
unsafe {
|
||||
std::ptr::drop_in_place(std::ptr::slice_from_raw_parts_mut(self.ptr.as_ptr(), len));
|
||||
if self.cap > 0 {
|
||||
let layout = Layout::array::<T>(self.cap).expect("layout");
|
||||
alloc::dealloc(self.ptr.as_ptr().cast(), layout);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl<T> StableVec<T> {
|
||||
pub fn from_vec_with_reserve(mut vec: Vec<T>, extra: usize) -> Self {
|
||||
vec.reserve(extra);
|
||||
let cap = vec.capacity();
|
||||
let len = vec.len();
|
||||
|
||||
let inner = if cap == 0 {
|
||||
StableBuf {
|
||||
ptr: NonNull::dangling(),
|
||||
cap: 0,
|
||||
len: AtomicUsize::new(0),
|
||||
}
|
||||
} else {
|
||||
// Take ownership of the Vec's buffer without running element
|
||||
// drops; we hand them off to the StableBuf.
|
||||
let mut vec = std::mem::ManuallyDrop::new(vec);
|
||||
let ptr = NonNull::new(vec.as_mut_ptr()).expect("non-null");
|
||||
StableBuf {
|
||||
ptr,
|
||||
cap,
|
||||
len: AtomicUsize::new(len),
|
||||
}
|
||||
};
|
||||
|
||||
Self {
|
||||
inner: Arc::new(inner),
|
||||
}
|
||||
}
|
||||
|
||||
/// Append. Returns `false` if capacity is exhausted (item dropped).
|
||||
///
|
||||
/// Safe to call via `&self` as long as the caller holds the outer
|
||||
/// picker write lock (single-writer invariant).
|
||||
#[inline]
|
||||
pub fn push(&self, item: T) -> bool {
|
||||
let cap = self.inner.cap;
|
||||
let len = self.inner.len.load(Ordering::Acquire);
|
||||
if len >= cap {
|
||||
debug_assert!(
|
||||
false,
|
||||
"StableVec: push would exceed capacity ({len} at capacity {cap})"
|
||||
);
|
||||
tracing::error!(
|
||||
len,
|
||||
capacity = cap,
|
||||
"StableVec: capacity exhausted — dropping item to prevent reallocation"
|
||||
);
|
||||
return false;
|
||||
}
|
||||
unsafe {
|
||||
std::ptr::write(self.inner.ptr.as_ptr().add(len), item);
|
||||
}
|
||||
self.inner.len.store(len + 1, Ordering::Release);
|
||||
true
|
||||
}
|
||||
|
||||
// this method is specifically private because you probably need to use
|
||||
// live_count if you are trying to access this method
|
||||
#[inline]
|
||||
pub fn len(&self) -> usize {
|
||||
self.inner.len.load(Ordering::Acquire)
|
||||
}
|
||||
|
||||
/// Mutable element access for in-place field updates. Never shifts.
|
||||
///
|
||||
/// LATENT UB: produces `&mut T` aliasing Arc-shared memory; the
|
||||
/// `&mut self` on StableVec does NOT imply unique access to the
|
||||
/// `StableBuf` when sibling Arc clones exist. Safe in practice
|
||||
/// because callers hold the picker write lock and writes target
|
||||
/// disjoint fields, but strictly forbidden by the aliasing model.
|
||||
#[inline]
|
||||
pub fn get_mut(&mut self, index: usize) -> Option<&mut T> {
|
||||
let len = self.inner.len.load(Ordering::Acquire);
|
||||
if index >= len {
|
||||
return None;
|
||||
}
|
||||
unsafe { Some(&mut *self.inner.ptr.as_ptr().add(index)) }
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub fn last(&self) -> Option<&T> {
|
||||
let len = self.len();
|
||||
if len == 0 {
|
||||
None
|
||||
} else {
|
||||
unsafe { Some(&*self.inner.ptr.as_ptr().add(len - 1)) }
|
||||
}
|
||||
}
|
||||
|
||||
/// Iterate mutably for in-place field updates. Never shifts storage.
|
||||
/// Same latent-UB caveat as [`get_mut`]: `&mut T` into Arc-shared memory.
|
||||
#[inline]
|
||||
pub fn iter_mut(&mut self) -> std::slice::IterMut<'_, T> {
|
||||
let len = self.inner.len.load(Ordering::Acquire);
|
||||
unsafe { std::slice::from_raw_parts_mut(self.inner.ptr.as_ptr(), len).iter_mut() }
|
||||
}
|
||||
}
|
||||
|
||||
impl<T> Clone for StableVec<T> {
|
||||
#[inline]
|
||||
fn clone(&self) -> Self {
|
||||
Self {
|
||||
inner: Arc::clone(&self.inner),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl<T: std::fmt::Debug> std::fmt::Debug for StableVec<T> {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
f.debug_tuple("StableVec").field(&self.len()).finish()
|
||||
}
|
||||
}
|
||||
|
||||
impl<T> std::ops::Deref for StableVec<T> {
|
||||
type Target = [T];
|
||||
#[inline]
|
||||
fn deref(&self) -> &[T] {
|
||||
let len = self.len();
|
||||
unsafe { std::slice::from_raw_parts(self.inner.ptr.as_ptr(), len) }
|
||||
}
|
||||
}
|
||||
|
||||
impl<T> std::ops::DerefMut for StableVec<T> {
|
||||
/// LATENT UB: `&mut [T]` aliases Arc-shared memory. Kept for
|
||||
/// Index/IndexMut ergonomics at call sites that write disjoint
|
||||
/// fields under the picker write lock. See module-level doc.
|
||||
#[inline]
|
||||
fn deref_mut(&mut self) -> &mut [T] {
|
||||
let len = self.inner.len.load(Ordering::Acquire);
|
||||
unsafe { std::slice::from_raw_parts_mut(self.inner.ptr.as_ptr(), len) }
|
||||
}
|
||||
}
|
||||
+800
-103
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,410 @@
|
||||
//! Integration test: verify that modifying a file after the bigram index is built
|
||||
//! still makes the new content findable via grep (through the overlay layer).
|
||||
|
||||
use std::fs;
|
||||
use std::time::Duration;
|
||||
use tempfile::TempDir;
|
||||
|
||||
use fff_search::file_picker::{FFFMode, FilePicker};
|
||||
use fff_search::grep::{GrepMode, GrepSearchOptions, parse_grep_query};
|
||||
use fff_search::{FilePickerOptions, SharedFilePicker, SharedFrecency};
|
||||
|
||||
/// Create a temp directory with some initial files, run the full picker lifecycle,
|
||||
/// then modify a file and verify grep finds the new content.
|
||||
#[test]
|
||||
fn modified_file_findable_via_overlay() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let base = tmp.path();
|
||||
|
||||
// Create initial files with known content.
|
||||
fs::write(base.join("alpha.txt"), "hello world\nfoo bar\n").unwrap();
|
||||
fs::write(
|
||||
base.join("beta.txt"),
|
||||
"some other content\nnothing special\n",
|
||||
)
|
||||
.unwrap();
|
||||
fs::write(base.join("gamma.txt"), "yet another file\nmore lines\n").unwrap();
|
||||
|
||||
let shared_picker = SharedFilePicker::default();
|
||||
let shared_frecency = SharedFrecency::default();
|
||||
|
||||
FilePicker::new_with_shared_state(
|
||||
shared_picker.clone(),
|
||||
shared_frecency.clone(),
|
||||
FilePickerOptions {
|
||||
base_path: base.to_string_lossy().to_string(),
|
||||
enable_mmap_cache: true,
|
||||
enable_content_indexing: true,
|
||||
mode: FFFMode::Neovim,
|
||||
watch: false, // we drive events manually
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.expect("Failed to create FilePicker");
|
||||
|
||||
// Wait for scan + bigram build to complete.
|
||||
let deadline = std::time::Instant::now() + Duration::from_secs(30);
|
||||
loop {
|
||||
std::thread::sleep(Duration::from_millis(50));
|
||||
|
||||
let ready = shared_picker
|
||||
.read()
|
||||
.ok()
|
||||
.map(|guard| {
|
||||
guard
|
||||
.as_ref()
|
||||
.map_or(false, |p| !p.is_scan_active() && p.bigram_index().is_some())
|
||||
})
|
||||
.unwrap_or(false);
|
||||
|
||||
if ready {
|
||||
break;
|
||||
}
|
||||
assert!(
|
||||
std::time::Instant::now() < deadline,
|
||||
"Timed out waiting for scan + bigram build"
|
||||
);
|
||||
}
|
||||
|
||||
// Sanity check: the 3 files are indexed.
|
||||
{
|
||||
let guard = shared_picker.read().unwrap();
|
||||
let picker = guard.as_ref().unwrap();
|
||||
assert_eq!(picker.get_files().len(), 3, "Expected 3 files after scan");
|
||||
assert!(
|
||||
picker.bigram_index().is_some(),
|
||||
"Bigram index should be built"
|
||||
);
|
||||
assert!(
|
||||
picker.bigram_overlay().is_some(),
|
||||
"Overlay should be initialized"
|
||||
);
|
||||
}
|
||||
|
||||
// "UNIQUE_NEEDLE" should NOT exist in any file yet.
|
||||
{
|
||||
let guard = shared_picker.read().unwrap();
|
||||
let picker = guard.as_ref().unwrap();
|
||||
let parsed = parse_grep_query("UNIQUE_NEEDLE");
|
||||
let opts = grep_opts();
|
||||
let result = picker.grep(&parsed, &opts);
|
||||
assert_eq!(
|
||||
result.matches.len(),
|
||||
0,
|
||||
"UNIQUE_NEEDLE should not exist before modification"
|
||||
);
|
||||
}
|
||||
|
||||
// Sleep so the filesystem mtime (seconds granularity) advances past the
|
||||
// value recorded during scan. Without this, on_create_or_modify skips
|
||||
// mmap invalidation and grep reads stale cached content.
|
||||
std::thread::sleep(Duration::from_millis(1100));
|
||||
|
||||
// Write new content containing the needle.
|
||||
let modified_path = base.join("beta.txt");
|
||||
fs::write(
|
||||
&modified_path,
|
||||
"some other content\nUNIQUE_NEEDLE is here\nnothing special\n",
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
// Simulate watcher event: call on_create_or_modify.
|
||||
// This updates the overlay's bigrams and invalidates the mmap cache.
|
||||
{
|
||||
let mut guard = shared_picker.write().unwrap();
|
||||
let picker = guard.as_mut().unwrap();
|
||||
let result = picker.handle_create_or_modify(&modified_path);
|
||||
assert!(
|
||||
result.is_some(),
|
||||
"on_create_or_modify should return the file"
|
||||
);
|
||||
}
|
||||
|
||||
// The bigram index was built BEFORE the modification, so without the
|
||||
// overlay, beta.txt would be filtered out (its old bigrams don't contain
|
||||
// "UNIQUE_NEEDLE"). The overlay should fix that.
|
||||
{
|
||||
let guard = shared_picker.read().unwrap();
|
||||
let picker = guard.as_ref().unwrap();
|
||||
let parsed = parse_grep_query("UNIQUE_NEEDLE");
|
||||
let opts = grep_opts();
|
||||
let result = picker.grep(&parsed, &opts);
|
||||
assert!(
|
||||
!result.matches.is_empty(),
|
||||
"UNIQUE_NEEDLE should be findable after modification (overlay adds the candidate back)"
|
||||
);
|
||||
// May find 1 or 2 matches depending on mmap cache state — the important
|
||||
// thing is that the modified content IS found.
|
||||
assert!(
|
||||
result
|
||||
.matches
|
||||
.iter()
|
||||
.any(|m| m.line_content.contains("UNIQUE_NEEDLE")),
|
||||
"At least one match should contain UNIQUE_NEEDLE"
|
||||
);
|
||||
}
|
||||
|
||||
// Cleanup: stop background watcher.
|
||||
if let Ok(mut guard) = shared_picker.write() {
|
||||
if let Some(ref mut picker) = *guard {
|
||||
picker.stop_background_monitor();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Verify that deleting a file makes its content un-findable via grep.
|
||||
#[test]
|
||||
fn deleted_file_excluded_via_overlay() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let base = tmp.path();
|
||||
|
||||
fs::write(base.join("keep.txt"), "keep this content\n").unwrap();
|
||||
fs::write(base.join("remove.txt"), "DELETEME_TOKEN is here\n").unwrap();
|
||||
|
||||
let shared_picker = SharedFilePicker::default();
|
||||
let shared_frecency = SharedFrecency::default();
|
||||
|
||||
FilePicker::new_with_shared_state(
|
||||
shared_picker.clone(),
|
||||
shared_frecency.clone(),
|
||||
FilePickerOptions {
|
||||
base_path: base.to_string_lossy().to_string(),
|
||||
enable_mmap_cache: true,
|
||||
enable_content_indexing: true,
|
||||
mode: FFFMode::Neovim,
|
||||
watch: false, // we drive events manually
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
wait_for_bigram(&shared_picker);
|
||||
|
||||
// Sanity: DELETEME_TOKEN is findable.
|
||||
{
|
||||
let guard = shared_picker.read().unwrap();
|
||||
let picker = guard.as_ref().unwrap();
|
||||
let result = grep_for(picker, "DELETEME_TOKEN");
|
||||
assert_eq!(
|
||||
result.matches.len(),
|
||||
1,
|
||||
"Token should be found before delete"
|
||||
);
|
||||
}
|
||||
|
||||
// Delete the file on disk and via picker.
|
||||
let remove_path = base.join("remove.txt");
|
||||
fs::remove_file(&remove_path).unwrap();
|
||||
{
|
||||
let mut guard = shared_picker.write().unwrap();
|
||||
let picker = guard.as_mut().unwrap();
|
||||
assert!(
|
||||
picker.remove_file_by_path(&remove_path),
|
||||
"remove should succeed"
|
||||
);
|
||||
}
|
||||
|
||||
// Token should no longer be found (tombstone in overlay clears the candidate).
|
||||
{
|
||||
let guard = shared_picker.read().unwrap();
|
||||
let picker = guard.as_ref().unwrap();
|
||||
let result = grep_for(picker, "DELETEME_TOKEN");
|
||||
assert_eq!(
|
||||
result.matches.len(),
|
||||
0,
|
||||
"DELETEME_TOKEN should not be found after deletion (tombstone in overlay)"
|
||||
);
|
||||
}
|
||||
|
||||
if let Ok(mut guard) = shared_picker.write() {
|
||||
if let Some(ref mut picker) = *guard {
|
||||
picker.stop_background_monitor();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Verify that a newly added file (in overflow) is findable via grep.
|
||||
#[test]
|
||||
fn new_file_findable_after_add() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let base = tmp.path();
|
||||
|
||||
fs::write(base.join("existing.txt"), "original content\n").unwrap();
|
||||
|
||||
let shared_picker = SharedFilePicker::default();
|
||||
let shared_frecency = SharedFrecency::default();
|
||||
|
||||
FilePicker::new_with_shared_state(
|
||||
shared_picker.clone(),
|
||||
shared_frecency.clone(),
|
||||
FilePickerOptions {
|
||||
base_path: base.to_string_lossy().to_string(),
|
||||
enable_mmap_cache: true,
|
||||
enable_content_indexing: true,
|
||||
mode: FFFMode::Neovim,
|
||||
watch: false, // we drive events manually
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
wait_for_bigram(&shared_picker);
|
||||
|
||||
// Create a new file on disk after the index was built.
|
||||
let new_path = base.join("newcomer.txt");
|
||||
fs::write(&new_path, "BRAND_NEW_TOKEN lives here\n").unwrap();
|
||||
|
||||
// Simulate watcher detecting the new file.
|
||||
{
|
||||
let mut guard = shared_picker.write().unwrap();
|
||||
let picker = guard.as_mut().unwrap();
|
||||
let result = picker.handle_create_or_modify(&new_path);
|
||||
assert!(
|
||||
result.is_some(),
|
||||
"on_create_or_modify should return the new file"
|
||||
);
|
||||
}
|
||||
|
||||
// The new file is in overflow, not in the base files slice.
|
||||
// grep_search currently only searches base files, so we need to verify
|
||||
// the overflow file is accessible.
|
||||
{
|
||||
let guard = shared_picker.read().unwrap();
|
||||
let picker = guard.as_ref().unwrap();
|
||||
let overflow = picker.get_overflow_files();
|
||||
assert_eq!(overflow.len(), 1, "Should have 1 overflow file");
|
||||
assert!(
|
||||
overflow[0].relative_path(picker).ends_with("newcomer.txt"),
|
||||
"Overflow file should be newcomer.txt"
|
||||
);
|
||||
}
|
||||
|
||||
if let Ok(mut guard) = shared_picker.write() {
|
||||
if let Some(ref mut picker) = *guard {
|
||||
picker.stop_background_monitor();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Verify that a file modified after index build is findable via regex grep
|
||||
/// through the overlay. This catches a regression where `extract_bigrams` on
|
||||
/// the raw regex string (e.g. "NEEDLE.*HERE") produces bogus bigrams containing
|
||||
/// `.` and `*`, causing `query_modified` to miss the file.
|
||||
#[test]
|
||||
fn modified_file_findable_via_regex_overlay() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let base = tmp.path();
|
||||
|
||||
fs::write(base.join("alpha.txt"), "hello world\nfoo bar\n").unwrap();
|
||||
fs::write(
|
||||
base.join("beta.txt"),
|
||||
"some other content\nnothing special\n",
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
let shared_picker = SharedFilePicker::default();
|
||||
let shared_frecency = SharedFrecency::default();
|
||||
|
||||
FilePicker::new_with_shared_state(
|
||||
shared_picker.clone(),
|
||||
shared_frecency.clone(),
|
||||
FilePickerOptions {
|
||||
base_path: base.to_string_lossy().to_string(),
|
||||
enable_mmap_cache: true,
|
||||
enable_content_indexing: true,
|
||||
mode: FFFMode::Neovim,
|
||||
watch: false, // we drive events manually
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
wait_for_bigram(&shared_picker);
|
||||
|
||||
// Advance mtime past the scan timestamp so the cache is invalidated.
|
||||
std::thread::sleep(Duration::from_millis(1100));
|
||||
|
||||
// Write content that matches the regex "NEEDLE.*HERE" into beta.txt.
|
||||
let modified_path = base.join("beta.txt");
|
||||
fs::write(
|
||||
&modified_path,
|
||||
"some other content\nNEEDLE is right HERE\nnothing special\n",
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
{
|
||||
let mut guard = shared_picker.write().unwrap();
|
||||
let picker = guard.as_mut().unwrap();
|
||||
assert!(picker.handle_create_or_modify(&modified_path).is_some());
|
||||
}
|
||||
|
||||
// Regex grep should find the modified file through the overlay.
|
||||
{
|
||||
let guard = shared_picker.read().unwrap();
|
||||
let picker = guard.as_ref().unwrap();
|
||||
let parsed = parse_grep_query("NEEDLE.*HERE");
|
||||
let opts = GrepSearchOptions {
|
||||
mode: GrepMode::Regex,
|
||||
..grep_opts()
|
||||
};
|
||||
let result = picker.grep(&parsed, &opts);
|
||||
assert!(
|
||||
!result.matches.is_empty(),
|
||||
"Regex grep should find NEEDLE.*HERE in modified file via overlay"
|
||||
);
|
||||
assert!(result.matches[0].line_content.contains("NEEDLE"));
|
||||
}
|
||||
|
||||
if let Ok(mut guard) = shared_picker.write() {
|
||||
if let Some(ref mut picker) = *guard {
|
||||
picker.stop_background_monitor();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ── Helpers ─────────────────────────────────────────────────────────────
|
||||
|
||||
fn grep_opts() -> GrepSearchOptions {
|
||||
GrepSearchOptions {
|
||||
max_file_size: 10 * 1024 * 1024,
|
||||
max_matches_per_file: 200,
|
||||
smart_case: true,
|
||||
file_offset: 0,
|
||||
page_limit: 200,
|
||||
mode: GrepMode::PlainText,
|
||||
time_budget_ms: 0,
|
||||
before_context: 0,
|
||||
after_context: 0,
|
||||
classify_definitions: false,
|
||||
trim_whitespace: false,
|
||||
abort_signal: None,
|
||||
}
|
||||
}
|
||||
|
||||
fn grep_for<'a>(picker: &'a FilePicker, query: &str) -> fff_search::grep::GrepResult<'a> {
|
||||
let parsed = parse_grep_query(query);
|
||||
picker.grep(&parsed, &grep_opts())
|
||||
}
|
||||
|
||||
fn wait_for_bigram(shared_picker: &SharedFilePicker) {
|
||||
let deadline = std::time::Instant::now() + Duration::from_secs(30);
|
||||
loop {
|
||||
std::thread::sleep(Duration::from_millis(50));
|
||||
let ready = shared_picker
|
||||
.read()
|
||||
.ok()
|
||||
.map(|guard| {
|
||||
guard
|
||||
.as_ref()
|
||||
.map_or(false, |p| !p.is_scan_active() && p.bigram_index().is_some())
|
||||
})
|
||||
.unwrap_or(false);
|
||||
if ready {
|
||||
break;
|
||||
}
|
||||
assert!(
|
||||
std::time::Instant::now() < deadline,
|
||||
"Timed out waiting for bigram build"
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,161 @@
|
||||
// Regression pinning: dropping a picker during poset scan off-lock time
|
||||
use std::fs;
|
||||
use std::path::Path;
|
||||
use std::process::Command;
|
||||
use std::sync::Arc;
|
||||
use std::sync::atomic::{AtomicBool, Ordering};
|
||||
use std::time::Duration;
|
||||
use tempfile::TempDir;
|
||||
|
||||
use fff_search::file_picker::{FFFMode, FilePicker, FuzzySearchOptions};
|
||||
use fff_search::{FilePickerOptions, QueryParser, SharedFilePicker, SharedFrecency};
|
||||
|
||||
fn seed_files(dir: &Path, count: usize) {
|
||||
for i in 0..count {
|
||||
let subdir = dir.join(format!("dir_{}", i / 20));
|
||||
fs::create_dir_all(&subdir).unwrap();
|
||||
fs::write(
|
||||
subdir.join(format!("file_{i}.rs")),
|
||||
format!("pub fn func_{i}() {{ /* token_{i} */ }}\n"),
|
||||
)
|
||||
.unwrap();
|
||||
}
|
||||
}
|
||||
|
||||
fn git_init(dir: &Path) {
|
||||
let run = |args: &[&str]| {
|
||||
Command::new("git")
|
||||
.args(args)
|
||||
.current_dir(dir)
|
||||
.env("GIT_AUTHOR_NAME", "test")
|
||||
.env("GIT_AUTHOR_EMAIL", "t@t")
|
||||
.env("GIT_COMMITTER_NAME", "test")
|
||||
.env("GIT_COMMITTER_EMAIL", "t@t")
|
||||
.output()
|
||||
.unwrap();
|
||||
};
|
||||
run(&["init"]);
|
||||
run(&["add", "-A"]);
|
||||
run(&["commit", "-m", "init"]);
|
||||
}
|
||||
|
||||
fn make_picker(base: &Path) -> (SharedFilePicker, SharedFrecency) {
|
||||
let sp = SharedFilePicker::default();
|
||||
let sf = SharedFrecency::default();
|
||||
FilePicker::new_with_shared_state(
|
||||
sp.clone(),
|
||||
sf.clone(),
|
||||
FilePickerOptions {
|
||||
base_path: base.to_string_lossy().to_string(),
|
||||
enable_mmap_cache: true,
|
||||
enable_content_indexing: true,
|
||||
mode: FFFMode::Neovim,
|
||||
watch: false,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.expect("init");
|
||||
(sp, sf)
|
||||
}
|
||||
|
||||
/// Drop picker immediately after scan starts — scan thread will find
|
||||
/// the picker gone and exit cleanly.
|
||||
#[test]
|
||||
fn drop_picker_during_walk_no_segfault() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
seed_files(tmp.path(), 500);
|
||||
git_init(tmp.path());
|
||||
|
||||
let (sp, _sf) = make_picker(tmp.path());
|
||||
// Don't wait — drop immediately while walk is likely in progress
|
||||
drop(sp);
|
||||
|
||||
// If we get here without SIGSEGV, the test passes.
|
||||
std::thread::sleep(Duration::from_millis(200));
|
||||
}
|
||||
|
||||
/// Drop picker while post-scan indexing is running. The snapshot holds
|
||||
/// Arc clones that keep the buffers alive.
|
||||
#[test]
|
||||
fn drop_picker_during_post_scan_no_segfault() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
seed_files(tmp.path(), 500);
|
||||
git_init(tmp.path());
|
||||
|
||||
let (sp, _sf) = make_picker(tmp.path());
|
||||
|
||||
// Wait for walk to finish (files are searchable) but post-scan is
|
||||
// still running (bigram not yet built).
|
||||
sp.wait_for_scan(Duration::from_secs(10));
|
||||
|
||||
// At this point post_scan_indexing_active is likely true.
|
||||
// Drop the picker — this releases the picker's Arc clones, but the
|
||||
// post-scan snapshot's clones keep the buffers alive.
|
||||
if let Ok(mut guard) = sp.write() {
|
||||
guard.take(); // drop the FilePicker
|
||||
}
|
||||
|
||||
// Give post-scan threads time to run against the "dead" picker.
|
||||
// They must not segfault.
|
||||
std::thread::sleep(Duration::from_secs(2));
|
||||
}
|
||||
|
||||
/// Drop picker from a second thread while the first thread is doing
|
||||
/// fuzzy searches. Verifies no segfault from interleaved access.
|
||||
#[test]
|
||||
fn drop_picker_concurrent_with_search_no_segfault() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
seed_files(tmp.path(), 500);
|
||||
git_init(tmp.path());
|
||||
|
||||
let (sp, _sf) = make_picker(tmp.path());
|
||||
sp.wait_for_scan(Duration::from_secs(10));
|
||||
|
||||
let sp_clone = sp.clone();
|
||||
let running = Arc::new(AtomicBool::new(true));
|
||||
let running_clone = running.clone();
|
||||
|
||||
// Searcher thread: continuously queries while the picker lives
|
||||
let searcher = std::thread::spawn(move || {
|
||||
let parser = QueryParser::default();
|
||||
while running_clone.load(Ordering::Relaxed) {
|
||||
if let Ok(guard) = sp_clone.read() {
|
||||
if let Some(picker) = guard.as_ref() {
|
||||
let query = parser.parse("func");
|
||||
let _ = picker.fuzzy_search(&query, None, FuzzySearchOptions::default());
|
||||
}
|
||||
}
|
||||
std::thread::sleep(Duration::from_millis(1));
|
||||
}
|
||||
});
|
||||
|
||||
// Let searches run for a bit, then drop
|
||||
std::thread::sleep(Duration::from_millis(100));
|
||||
if let Ok(mut guard) = sp.write() {
|
||||
guard.take();
|
||||
}
|
||||
std::thread::sleep(Duration::from_millis(100));
|
||||
|
||||
running.store(false, Ordering::Relaxed);
|
||||
searcher.join().unwrap();
|
||||
}
|
||||
|
||||
/// Repeated init + wait + clean-drop cycle. This is the pattern that
|
||||
/// SIGSEGV'd on the pre-refactor code in the benchmark.
|
||||
#[test]
|
||||
fn repeated_init_and_drop_no_segfault() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
seed_files(tmp.path(), 200);
|
||||
git_init(tmp.path());
|
||||
|
||||
for _ in 0..5 {
|
||||
let (sp, _sf) = make_picker(tmp.path());
|
||||
sp.wait_for_scan(Duration::from_secs(10));
|
||||
sp.wait_for_indexing_complete(Duration::from_secs(30));
|
||||
if let Ok(mut guard) = sp.write()
|
||||
&& let Some(mut picker) = guard.take()
|
||||
{
|
||||
picker.stop_background_monitor();
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,320 @@
|
||||
//! Reproducer: macOS FSEvents does not deliver Remove events for files
|
||||
//! deleted from NonRecursive-watched directories when multiple directories
|
||||
//! are watched via stop/restart cycles.
|
||||
//!
|
||||
//! This test watches a temp directory NonRecursively, creates a file,
|
||||
//! verifies the Create event, deletes the file, and checks whether a
|
||||
//! Remove (or any) event is delivered.
|
||||
|
||||
use notify::event::*;
|
||||
use notify::{Config, EventKindMask, RecommendedWatcher, RecursiveMode, Watcher};
|
||||
use std::fs;
|
||||
use std::path::PathBuf;
|
||||
use std::sync::mpsc;
|
||||
use std::time::Duration;
|
||||
|
||||
fn setup_temp_git_repo() -> (PathBuf, tempfile::TempDir) {
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
let dir = tmp.path().canonicalize().unwrap();
|
||||
|
||||
// Create a git repo like the bun test does
|
||||
std::process::Command::new("git")
|
||||
.args(["init", "-b", "main"])
|
||||
.current_dir(&dir)
|
||||
.output()
|
||||
.unwrap();
|
||||
|
||||
fs::write(dir.join("hello.txt"), "hello\n").unwrap();
|
||||
fs::create_dir_all(dir.join("src")).unwrap();
|
||||
fs::write(dir.join("src/main.rs"), "fn main() {}\n").unwrap();
|
||||
|
||||
std::process::Command::new("git")
|
||||
.args(["add", "-A"])
|
||||
.current_dir(&dir)
|
||||
.output()
|
||||
.unwrap();
|
||||
std::process::Command::new("git")
|
||||
.args(["commit", "-m", "init"])
|
||||
.current_dir(&dir)
|
||||
.output()
|
||||
.unwrap();
|
||||
|
||||
(dir, tmp)
|
||||
}
|
||||
|
||||
/// Raw notify watcher: single NonRecursive watch on a directory.
|
||||
/// Create a file, delete it, check if Remove event is delivered.
|
||||
#[test]
|
||||
fn raw_notify_nonrecursive_detects_deletion() {
|
||||
let (dir, _tmp) = setup_temp_git_repo();
|
||||
let (tx, rx) = mpsc::channel();
|
||||
|
||||
let config = Config::default()
|
||||
.with_follow_symlinks(false)
|
||||
.with_event_kinds(EventKindMask::CORE);
|
||||
|
||||
let mut watcher = RecommendedWatcher::new(
|
||||
move |res: notify::Result<Event>| {
|
||||
if let Ok(ev) = res {
|
||||
let _ = tx.send(ev);
|
||||
}
|
||||
},
|
||||
config,
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
// Watch ONLY the root dir NonRecursively (like fff does)
|
||||
watcher
|
||||
.watch(dir.as_path(), RecursiveMode::NonRecursive)
|
||||
.unwrap();
|
||||
|
||||
// Let the watcher stabilize
|
||||
std::thread::sleep(Duration::from_millis(500));
|
||||
|
||||
// Drain any startup events
|
||||
while rx.try_recv().is_ok() {}
|
||||
|
||||
// Create a file
|
||||
let file_path = dir.join("testfile.txt");
|
||||
fs::write(&file_path, "content\n").unwrap();
|
||||
|
||||
// Wait for create event
|
||||
let mut got_create = false;
|
||||
let deadline = std::time::Instant::now() + Duration::from_secs(5);
|
||||
while std::time::Instant::now() < deadline {
|
||||
match rx.recv_timeout(Duration::from_millis(100)) {
|
||||
Ok(ev) => {
|
||||
eprintln!(" [create phase] event: {:?} paths={:?}", ev.kind, ev.paths);
|
||||
if matches!(ev.kind, EventKind::Create(_)) && ev.paths.contains(&file_path) {
|
||||
got_create = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
Err(mpsc::RecvTimeoutError::Timeout) => continue,
|
||||
Err(_) => break,
|
||||
}
|
||||
}
|
||||
assert!(got_create, "Expected Create event for testfile.txt");
|
||||
|
||||
// Drain remaining events from the create
|
||||
std::thread::sleep(Duration::from_millis(300));
|
||||
while rx.try_recv().is_ok() {}
|
||||
|
||||
// Delete the file
|
||||
fs::remove_file(&file_path).unwrap();
|
||||
eprintln!(" File deleted: {}", file_path.display());
|
||||
|
||||
// Wait for any event related to the deletion
|
||||
let mut got_removal_event = false;
|
||||
let deadline = std::time::Instant::now() + Duration::from_secs(5);
|
||||
while std::time::Instant::now() < deadline {
|
||||
match rx.recv_timeout(Duration::from_millis(100)) {
|
||||
Ok(ev) => {
|
||||
eprintln!(" [delete phase] event: {:?} paths={:?}", ev.kind, ev.paths);
|
||||
if ev.paths.contains(&file_path) {
|
||||
got_removal_event = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
Err(mpsc::RecvTimeoutError::Timeout) => continue,
|
||||
Err(_) => break,
|
||||
}
|
||||
}
|
||||
|
||||
assert!(
|
||||
got_removal_event,
|
||||
"Expected some event for deleted testfile.txt but got none within 5s"
|
||||
);
|
||||
}
|
||||
|
||||
/// Same test but with MULTIPLE NonRecursive watches (base + src + .git)
|
||||
/// to match what fff actually does. Each watch() call stops/restarts the FSEvents stream.
|
||||
#[test]
|
||||
fn raw_notify_multi_nonrecursive_detects_deletion() {
|
||||
let (dir, _tmp) = setup_temp_git_repo();
|
||||
let (tx, rx) = mpsc::channel();
|
||||
|
||||
let config = Config::default()
|
||||
.with_follow_symlinks(false)
|
||||
.with_event_kinds(EventKindMask::CORE);
|
||||
|
||||
let mut watcher = RecommendedWatcher::new(
|
||||
move |res: notify::Result<Event>| {
|
||||
if let Ok(ev) = res {
|
||||
let _ = tx.send(ev);
|
||||
}
|
||||
},
|
||||
config,
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
// Watch multiple directories NonRecursively — EACH call restarts the FSEvents stream
|
||||
watcher
|
||||
.watch(dir.as_path(), RecursiveMode::NonRecursive)
|
||||
.unwrap();
|
||||
watcher
|
||||
.watch(&dir.join("src"), RecursiveMode::NonRecursive)
|
||||
.unwrap();
|
||||
watcher
|
||||
.watch(&dir.join(".git"), RecursiveMode::NonRecursive)
|
||||
.unwrap();
|
||||
|
||||
// Let the watcher stabilize
|
||||
std::thread::sleep(Duration::from_millis(500));
|
||||
while rx.try_recv().is_ok() {}
|
||||
|
||||
// Create a file in root dir
|
||||
let file_path = dir.join("testfile.txt");
|
||||
fs::write(&file_path, "content\n").unwrap();
|
||||
|
||||
// Wait for create event
|
||||
let mut got_create = false;
|
||||
let deadline = std::time::Instant::now() + Duration::from_secs(5);
|
||||
while std::time::Instant::now() < deadline {
|
||||
match rx.recv_timeout(Duration::from_millis(100)) {
|
||||
Ok(ev) => {
|
||||
eprintln!(" [multi-create] event: {:?} paths={:?}", ev.kind, ev.paths);
|
||||
if matches!(ev.kind, EventKind::Create(_)) && ev.paths.contains(&file_path) {
|
||||
got_create = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
Err(mpsc::RecvTimeoutError::Timeout) => continue,
|
||||
Err(_) => break,
|
||||
}
|
||||
}
|
||||
assert!(
|
||||
got_create,
|
||||
"Expected Create event for testfile.txt with multi-watch"
|
||||
);
|
||||
|
||||
// Drain
|
||||
std::thread::sleep(Duration::from_millis(300));
|
||||
while rx.try_recv().is_ok() {}
|
||||
|
||||
// Delete the file
|
||||
fs::remove_file(&file_path).unwrap();
|
||||
eprintln!(" File deleted: {}", file_path.display());
|
||||
|
||||
// Wait for any event related to the deletion
|
||||
let mut got_removal_event = false;
|
||||
let deadline = std::time::Instant::now() + Duration::from_secs(5);
|
||||
while std::time::Instant::now() < deadline {
|
||||
match rx.recv_timeout(Duration::from_millis(100)) {
|
||||
Ok(ev) => {
|
||||
eprintln!(" [multi-delete] event: {:?} paths={:?}", ev.kind, ev.paths);
|
||||
if ev.paths.contains(&file_path) {
|
||||
got_removal_event = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
Err(mpsc::RecvTimeoutError::Timeout) => continue,
|
||||
Err(_) => break,
|
||||
}
|
||||
}
|
||||
|
||||
assert!(
|
||||
got_removal_event,
|
||||
"Expected some event for deleted testfile.txt with multi-watch but got none within 5s"
|
||||
);
|
||||
}
|
||||
|
||||
/// Test with debouncer (matching exactly what fff uses)
|
||||
#[test]
|
||||
fn debounced_nonrecursive_detects_deletion() {
|
||||
use notify_debouncer_full::{DebounceEventResult, NoCache, new_debouncer_opt};
|
||||
|
||||
let (dir, _tmp) = setup_temp_git_repo();
|
||||
let (tx, rx) = mpsc::channel();
|
||||
|
||||
let config = Config::default()
|
||||
.with_follow_symlinks(false)
|
||||
.with_event_kinds(EventKindMask::CORE);
|
||||
|
||||
let mut debouncer: notify_debouncer_full::Debouncer<RecommendedWatcher, NoCache> =
|
||||
new_debouncer_opt(
|
||||
Duration::from_millis(250),
|
||||
Some(Duration::from_millis(125)),
|
||||
move |result: DebounceEventResult| {
|
||||
if let Ok(events) = result {
|
||||
for ev in events {
|
||||
eprintln!(
|
||||
" [debounced-cb] kind={:?} paths={:?}",
|
||||
ev.event.kind, ev.event.paths
|
||||
);
|
||||
let _ = tx.send(ev);
|
||||
}
|
||||
}
|
||||
},
|
||||
NoCache::new(),
|
||||
config,
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
// Watch like fff does
|
||||
debouncer
|
||||
.watch(dir.as_path(), RecursiveMode::NonRecursive)
|
||||
.unwrap();
|
||||
debouncer
|
||||
.watch(&dir.join("src"), RecursiveMode::NonRecursive)
|
||||
.unwrap();
|
||||
debouncer
|
||||
.watch(&dir.join(".git"), RecursiveMode::NonRecursive)
|
||||
.unwrap();
|
||||
|
||||
// Longer stabilization — each watch() restarts the FSEvents stream
|
||||
std::thread::sleep(Duration::from_secs(1));
|
||||
while rx.try_recv().is_ok() {}
|
||||
|
||||
// Create file
|
||||
let file_path = dir.join("testfile.txt");
|
||||
fs::write(&file_path, "content\n").unwrap();
|
||||
|
||||
let mut got_create = false;
|
||||
let deadline = std::time::Instant::now() + Duration::from_secs(5);
|
||||
while std::time::Instant::now() < deadline {
|
||||
match rx.recv_timeout(Duration::from_millis(100)) {
|
||||
Ok(ev) => {
|
||||
if ev.event.paths.contains(&file_path) {
|
||||
got_create = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
Err(mpsc::RecvTimeoutError::Timeout) => continue,
|
||||
Err(_) => break,
|
||||
}
|
||||
}
|
||||
assert!(got_create, "Expected Create event via debouncer");
|
||||
|
||||
// Wait for debounce to fully flush
|
||||
std::thread::sleep(Duration::from_millis(500));
|
||||
while rx.try_recv().is_ok() {}
|
||||
|
||||
// Delete
|
||||
fs::remove_file(&file_path).unwrap();
|
||||
eprintln!(" File deleted: {}", file_path.display());
|
||||
|
||||
// Wait for ANY event for this path
|
||||
let mut got_event = false;
|
||||
let mut event_kind = String::new();
|
||||
let deadline = std::time::Instant::now() + Duration::from_secs(5);
|
||||
while std::time::Instant::now() < deadline {
|
||||
match rx.recv_timeout(Duration::from_millis(100)) {
|
||||
Ok(ev) => {
|
||||
if ev.event.paths.contains(&file_path) {
|
||||
event_kind = format!("{:?}", ev.event.kind);
|
||||
got_event = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
Err(mpsc::RecvTimeoutError::Timeout) => continue,
|
||||
Err(_) => break,
|
||||
}
|
||||
}
|
||||
|
||||
assert!(
|
||||
got_event,
|
||||
"Expected some event for deleted testfile.txt via debouncer but got none within 5s"
|
||||
);
|
||||
eprintln!(" Got event kind: {}", event_kind);
|
||||
}
|
||||
@@ -0,0 +1,860 @@
|
||||
//! Randomized file-system mutation stress test.
|
||||
//!
|
||||
//! Seeds a directory with ~40 files across diverse content domains, builds the
|
||||
//! picker + bigram index, then runs 20 rounds of randomized create / edit /
|
||||
//! delete / rename / read-only operations. After every round the test verifies
|
||||
//! that plain-text grep, regex grep, and fuzzy file search all return correct
|
||||
//! results for every live and dead file.
|
||||
//!
|
||||
//! Uses a seeded RNG (`SmallRng::seed_from_u64`) for deterministic
|
||||
//! reproduction.
|
||||
|
||||
use std::fs;
|
||||
use std::path::{Path, PathBuf};
|
||||
use std::process::Command;
|
||||
use std::time::{Duration, SystemTime, UNIX_EPOCH};
|
||||
use tempfile::TempDir;
|
||||
|
||||
use rand::rngs::SmallRng;
|
||||
use rand::{RngCore, SeedableRng};
|
||||
|
||||
use fff_search::file_picker::{FFFMode, FilePicker, FuzzySearchOptions};
|
||||
use fff_search::grep::{GrepMode, GrepSearchOptions, parse_grep_query};
|
||||
use fff_search::{
|
||||
FilePickerOptions, PaginationArgs, QueryParser, SharedFilePicker, SharedFrecency,
|
||||
};
|
||||
|
||||
const DOMAINS: &[&str] = &[
|
||||
r#"
|
||||
use std::net::{TcpStream, SocketAddr};
|
||||
fn establish_connection(addr: SocketAddr) -> Result<TcpStream, std::io::Error> {
|
||||
let stream = TcpStream::connect(addr)?;
|
||||
stream.set_nodelay(true)?;
|
||||
Ok(stream)
|
||||
}
|
||||
fn parse_http_header(raw: &[u8]) -> Option<(&str, &str)> {
|
||||
let line = std::str::from_utf8(raw).ok()?;
|
||||
let (key, val) = line.split_once(':')?;
|
||||
Some((key.trim(), val.trim()))
|
||||
}
|
||||
"#,
|
||||
r#"
|
||||
use sqlx::{PgPool, Row};
|
||||
async fn query_users(pool: &PgPool, limit: i64) -> Vec<String> {
|
||||
sqlx::query("SELECT name FROM users ORDER BY created_at DESC LIMIT $1")
|
||||
.bind(limit)
|
||||
.fetch_all(pool)
|
||||
.await
|
||||
.unwrap()
|
||||
.iter()
|
||||
.map(|row| row.get("name"))
|
||||
.collect()
|
||||
}
|
||||
async fn insert_record(pool: &PgPool, name: &str) -> i64 {
|
||||
sqlx::query_scalar("INSERT INTO records (name) VALUES ($1) RETURNING id")
|
||||
.bind(name)
|
||||
.fetch_one(pool)
|
||||
.await
|
||||
.unwrap()
|
||||
}
|
||||
"#,
|
||||
r#"
|
||||
fn verify_jwt_token(token: &str, secret: &[u8]) -> Result<Claims, AuthError> {
|
||||
let parts: Vec<&str> = token.splitn(3, '.').collect();
|
||||
if parts.len() != 3 { return Err(AuthError::MalformedToken); }
|
||||
let payload = base64_decode(parts[1])?;
|
||||
let signature = hmac_sha256(secret, &format!("{}.{}", parts[0], parts[1]));
|
||||
if signature != base64_decode(parts[2])? { return Err(AuthError::InvalidSignature); }
|
||||
serde_json::from_slice(&payload).map_err(AuthError::Deserialize)
|
||||
}
|
||||
fn hash_password(password: &str, salt: &[u8]) -> Vec<u8> {
|
||||
argon2::hash_encoded(password.as_bytes(), salt, &argon2::Config::default())
|
||||
.unwrap().into_bytes()
|
||||
}
|
||||
"#,
|
||||
r#"
|
||||
struct Renderer { framebuffer: Vec<u32>, width: usize, height: usize }
|
||||
impl Renderer {
|
||||
fn clear(&mut self, color: u32) { self.framebuffer.fill(color); }
|
||||
fn draw_pixel(&mut self, x: usize, y: usize, color: u32) {
|
||||
if x < self.width && y < self.height {
|
||||
self.framebuffer[y * self.width + x] = color;
|
||||
}
|
||||
}
|
||||
fn draw_line(&mut self, x0: i32, y0: i32, x1: i32, y1: i32, color: u32) {
|
||||
let dx = (x1 - x0).abs(); let dy = -(y1 - y0).abs();
|
||||
let mut err = dx + dy;
|
||||
let (mut cx, mut cy) = (x0, y0);
|
||||
loop {
|
||||
self.draw_pixel(cx as usize, cy as usize, color);
|
||||
if cx == x1 && cy == y1 { break; }
|
||||
let e2 = 2 * err;
|
||||
if e2 >= dy { err += dy; cx += if x0 < x1 { 1 } else { -1 }; }
|
||||
if e2 <= dx { err += dx; cy += if y0 < y1 { 1 } else { -1 }; }
|
||||
}
|
||||
}
|
||||
}
|
||||
"#,
|
||||
r#"
|
||||
use serde::{Serialize, Deserialize};
|
||||
#[derive(Serialize, Deserialize)]
|
||||
struct ConfigFile { log_level: String, max_retries: u32, timeout_ms: u64 }
|
||||
fn load_config(path: &std::path::Path) -> Result<ConfigFile, Box<dyn std::error::Error>> {
|
||||
let contents = std::fs::read_to_string(path)?;
|
||||
let config: ConfigFile = toml::from_str(&contents)?;
|
||||
Ok(config)
|
||||
}
|
||||
fn merge_configs(base: ConfigFile, overlay: ConfigFile) -> ConfigFile {
|
||||
ConfigFile {
|
||||
log_level: if overlay.log_level.is_empty() { base.log_level } else { overlay.log_level },
|
||||
max_retries: overlay.max_retries.max(base.max_retries),
|
||||
timeout_ms: overlay.timeout_ms.max(base.timeout_ms),
|
||||
}
|
||||
}
|
||||
"#,
|
||||
r#"
|
||||
struct PhysicsBody { position: [f64; 3], velocity: [f64; 3], mass: f64 }
|
||||
fn apply_gravity(bodies: &mut [PhysicsBody], dt: f64) {
|
||||
let gravity_constant = 6.674e-11;
|
||||
let len = bodies.len();
|
||||
let mut forces = vec![[0.0f64; 3]; len];
|
||||
for i in 0..len {
|
||||
for j in (i+1)..len {
|
||||
let dx = bodies[j].position[0] - bodies[i].position[0];
|
||||
let dy = bodies[j].position[1] - bodies[i].position[1];
|
||||
let dz = bodies[j].position[2] - bodies[i].position[2];
|
||||
let dist_sq = dx*dx + dy*dy + dz*dz;
|
||||
let force_mag = gravity_constant * bodies[i].mass * bodies[j].mass / dist_sq;
|
||||
let dist = dist_sq.sqrt();
|
||||
for k in 0..3 {
|
||||
let f = force_mag * [dx, dy, dz][k] / dist;
|
||||
forces[i][k] += f; forces[j][k] -= f;
|
||||
}
|
||||
}
|
||||
}
|
||||
for (body, force) in bodies.iter_mut().zip(forces.iter()) {
|
||||
for k in 0..3 {
|
||||
body.velocity[k] += force[k] / body.mass * dt;
|
||||
body.position[k] += body.velocity[k] * dt;
|
||||
}
|
||||
}
|
||||
}
|
||||
"#,
|
||||
r#"
|
||||
use std::collections::BTreeMap;
|
||||
struct CacheEntry<V> { value: V, frequency: u64, last_access: u64 }
|
||||
struct LFUCache<K: Ord, V> { map: BTreeMap<K, CacheEntry<V>>, capacity: usize, clock: u64 }
|
||||
impl<K: Ord, V> LFUCache<K, V> {
|
||||
fn new(capacity: usize) -> Self { Self { map: BTreeMap::new(), capacity, clock: 0 } }
|
||||
fn get(&mut self, key: &K) -> Option<&V> {
|
||||
self.clock += 1;
|
||||
let entry = self.map.get_mut(key)?;
|
||||
entry.frequency += 1;
|
||||
entry.last_access = self.clock;
|
||||
Some(&entry.value)
|
||||
}
|
||||
fn insert(&mut self, key: K, value: V) {
|
||||
self.clock += 1;
|
||||
if self.map.len() >= self.capacity { self.evict(); }
|
||||
self.map.insert(key, CacheEntry { value, frequency: 1, last_access: self.clock });
|
||||
}
|
||||
fn evict(&mut self) {
|
||||
if let Some(victim) = self.map.keys().min_by_key(|k| {
|
||||
let e = &self.map[*k]; (e.frequency, e.last_access)
|
||||
}).cloned() { self.map.remove(&victim); }
|
||||
}
|
||||
}
|
||||
"#,
|
||||
r#"
|
||||
fn tokenize_expression(input: &str) -> Vec<Token> {
|
||||
let mut tokens = Vec::new();
|
||||
let mut chars = input.chars().peekable();
|
||||
while let Some(&ch) = chars.peek() {
|
||||
match ch {
|
||||
'0'..='9' => {
|
||||
let mut num = String::new();
|
||||
while let Some(&d) = chars.peek() {
|
||||
if d.is_ascii_digit() || d == '.' { num.push(d); chars.next(); }
|
||||
else { break; }
|
||||
}
|
||||
tokens.push(Token::Number(num.parse().unwrap()));
|
||||
}
|
||||
'+' => { tokens.push(Token::Plus); chars.next(); }
|
||||
'-' => { tokens.push(Token::Minus); chars.next(); }
|
||||
'*' => { tokens.push(Token::Star); chars.next(); }
|
||||
'/' => { tokens.push(Token::Slash); chars.next(); }
|
||||
'(' => { tokens.push(Token::LParen); chars.next(); }
|
||||
')' => { tokens.push(Token::RParen); chars.next(); }
|
||||
_ if ch.is_whitespace() => { chars.next(); }
|
||||
_ => { chars.next(); }
|
||||
}
|
||||
}
|
||||
tokens
|
||||
}
|
||||
"#,
|
||||
r#"
|
||||
use std::sync::mpsc;
|
||||
use std::thread;
|
||||
fn parallel_map<T: Send + 'static, R: Send + 'static>(
|
||||
items: Vec<T>, num_threads: usize, f: fn(T) -> R
|
||||
) -> Vec<R> {
|
||||
let chunk_size = (items.len() + num_threads - 1) / num_threads;
|
||||
let (tx, rx) = mpsc::channel();
|
||||
let mut handles = Vec::new();
|
||||
for (chunk_idx, chunk) in items.into_iter().collect::<Vec<_>>()
|
||||
.chunks(chunk_size).enumerate()
|
||||
{
|
||||
let tx = tx.clone();
|
||||
let chunk = chunk.to_vec();
|
||||
handles.push(thread::spawn(move || {
|
||||
for (i, item) in chunk.into_iter().enumerate() {
|
||||
tx.send((chunk_idx * chunk_size + i, f(item))).unwrap();
|
||||
}
|
||||
}));
|
||||
}
|
||||
drop(tx);
|
||||
let mut results: Vec<Option<R>> = vec![None; handles.len() * chunk_size];
|
||||
for (idx, result) in rx { if idx < results.len() { results[idx] = Some(result); } }
|
||||
for h in handles { h.join().unwrap(); }
|
||||
results.into_iter().flatten().collect()
|
||||
}
|
||||
"#,
|
||||
r#"
|
||||
struct Compressor { window: Vec<u8>, window_size: usize }
|
||||
impl Compressor {
|
||||
fn new(window_size: usize) -> Self {
|
||||
Self { window: Vec::with_capacity(window_size), window_size }
|
||||
}
|
||||
fn find_longest_match(&self, data: &[u8], pos: usize) -> (usize, usize) {
|
||||
let mut best_offset = 0; let mut best_length = 0;
|
||||
let start = pos.saturating_sub(self.window_size);
|
||||
for offset in start..pos {
|
||||
let mut length = 0;
|
||||
while pos + length < data.len()
|
||||
&& data[offset + length] == data[pos + length]
|
||||
&& length < 258
|
||||
{ length += 1; }
|
||||
if length > best_length { best_offset = pos - offset; best_length = length; }
|
||||
}
|
||||
(best_offset, best_length)
|
||||
}
|
||||
fn compress(&mut self, data: &[u8]) -> Vec<u8> {
|
||||
let mut output = Vec::new();
|
||||
let mut pos = 0;
|
||||
while pos < data.len() {
|
||||
let (offset, length) = self.find_longest_match(data, pos);
|
||||
if length >= 3 {
|
||||
output.push(1); output.extend_from_slice(&(offset as u16).to_le_bytes());
|
||||
output.push(length as u8); pos += length;
|
||||
} else { output.push(0); output.push(data[pos]); pos += 1; }
|
||||
}
|
||||
output
|
||||
}
|
||||
}
|
||||
"#,
|
||||
];
|
||||
|
||||
struct FileState {
|
||||
name: String,
|
||||
token: String,
|
||||
#[allow(dead_code)]
|
||||
is_base: bool,
|
||||
/// Epoch second when this file was last written (used to detect same-second
|
||||
/// re-edits that wouldn't bump mtime and thus wouldn't invalidate the mmap).
|
||||
last_write_sec: u64,
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn fuzz_file_operations_stress() {
|
||||
const SEED: u64 = 0xDEAD_BEEF_CAFE_1234;
|
||||
const INITIAL_FILE_COUNT: usize = 40;
|
||||
const NUM_ROUNDS: usize = 20;
|
||||
|
||||
let mut rng = SmallRng::seed_from_u64(SEED);
|
||||
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let base = tmp.path();
|
||||
|
||||
// Timing accumulators.
|
||||
let mut t_sleep = Duration::ZERO;
|
||||
let mut t_git = Duration::ZERO;
|
||||
let mut t_bigram_wait = Duration::ZERO;
|
||||
let mut t_grep_plain = Duration::ZERO;
|
||||
let mut t_grep_regex = Duration::ZERO;
|
||||
let mut t_fuzzy = Duration::ZERO;
|
||||
let mut t_dead_check = Duration::ZERO;
|
||||
let grep_plain_calls = std::sync::atomic::AtomicUsize::new(0);
|
||||
let grep_regex_calls = std::sync::atomic::AtomicUsize::new(0);
|
||||
let fuzzy_calls = std::sync::atomic::AtomicUsize::new(0);
|
||||
let dead_calls = std::sync::atomic::AtomicUsize::new(0);
|
||||
let test_start = std::time::Instant::now();
|
||||
|
||||
let mut live_files: Vec<FileState> = Vec::with_capacity(INITIAL_FILE_COUNT + NUM_ROUNDS);
|
||||
let mut dead_tokens: Vec<String> = Vec::new();
|
||||
let mut next_file_id: usize = 0;
|
||||
|
||||
for i in 0..INITIAL_FILE_COUNT {
|
||||
let name = format!("seed_{i:04}.rs");
|
||||
let token = format!("FUZZ_SEED_{i:04}");
|
||||
write_diverse_file(base, &name, &token, i);
|
||||
live_files.push(FileState {
|
||||
name,
|
||||
token,
|
||||
is_base: true,
|
||||
last_write_sec: 0, // set before index build, doesn't matter
|
||||
});
|
||||
next_file_id += 1;
|
||||
}
|
||||
|
||||
let t0 = std::time::Instant::now();
|
||||
git_init_and_commit(base);
|
||||
t_git += t0.elapsed();
|
||||
|
||||
let shared_picker = SharedFilePicker::default();
|
||||
|
||||
FilePicker::new_with_shared_state(
|
||||
shared_picker.clone(),
|
||||
SharedFrecency::noop(),
|
||||
FilePickerOptions {
|
||||
watch: false, // we do not need the backgrodun monitor
|
||||
base_path: base.to_string_lossy().to_string(),
|
||||
enable_mmap_cache: true,
|
||||
enable_content_indexing: true,
|
||||
mode: FFFMode::Neovim,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.expect("Failed to create FilePicker");
|
||||
|
||||
let t0 = std::time::Instant::now();
|
||||
wait_for_bigram(&shared_picker);
|
||||
t_bigram_wait += t0.elapsed();
|
||||
|
||||
// Sanity: all initial tokens findable via plain grep.
|
||||
{
|
||||
let guard = shared_picker.read().unwrap();
|
||||
let picker = guard.as_ref().unwrap();
|
||||
for fs in &live_files {
|
||||
assert!(
|
||||
grep_plain_count(picker, &fs.token) >= 1,
|
||||
"initial sanity: plain grep should find token {} in {}",
|
||||
fs.token,
|
||||
fs.name
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
// Sleep so mtime advances past the scan snapshot timestamp.
|
||||
let t0 = std::time::Instant::now();
|
||||
std::thread::sleep(Duration::from_millis(1100));
|
||||
t_sleep += t0.elapsed();
|
||||
|
||||
let mut op_counter: usize = 0;
|
||||
|
||||
for round in 0..NUM_ROUNDS {
|
||||
let roll: u32 = rng.next_u32() % 100;
|
||||
|
||||
if roll < 40 && !live_files.is_empty() {
|
||||
// ── EDIT existing file (40%) ──
|
||||
let idx = rng.next_u32() as usize % live_files.len();
|
||||
|
||||
// on_create_or_modify uses mtime (seconds granularity) to decide
|
||||
// whether to invalidate the mmap cache. If we re-edit a file in
|
||||
// the same second it was last written, the mtime won't change and
|
||||
// the stale cached content will be returned. Sleep to advance mtime.
|
||||
let now_sec = epoch_secs();
|
||||
if live_files[idx].last_write_sec >= now_sec {
|
||||
let t0 = std::time::Instant::now();
|
||||
std::thread::sleep(Duration::from_millis(1100));
|
||||
t_sleep += t0.elapsed();
|
||||
}
|
||||
|
||||
let old_token = live_files[idx].token.clone();
|
||||
let new_token = format!("FUZZ_{round:02}_{op_counter:04}");
|
||||
let name = &live_files[idx].name;
|
||||
let domain_idx = rng.next_u32() as usize % DOMAINS.len();
|
||||
write_diverse_file_with_domain(base, name, &new_token, domain_idx);
|
||||
|
||||
{
|
||||
let mut guard = shared_picker.write().unwrap();
|
||||
let picker = guard.as_mut().unwrap();
|
||||
assert!(
|
||||
picker.handle_create_or_modify(base.join(name)).is_some(),
|
||||
"round {round}: on_create_or_modify({name}) should succeed for edit"
|
||||
);
|
||||
}
|
||||
|
||||
dead_tokens.push(old_token);
|
||||
live_files[idx].token = new_token;
|
||||
live_files[idx].last_write_sec = epoch_secs();
|
||||
op_counter += 1;
|
||||
} else if roll < 60 {
|
||||
// ── CREATE new file (20%) ──
|
||||
let name = format!("created_{next_file_id:04}.rs");
|
||||
let token = format!("FUZZ_{round:02}_{op_counter:04}");
|
||||
let domain_idx = rng.next_u32() as usize % DOMAINS.len();
|
||||
write_diverse_file_with_domain(base, &name, &token, domain_idx);
|
||||
|
||||
{
|
||||
let mut guard = shared_picker.write().unwrap();
|
||||
let picker = guard.as_mut().unwrap();
|
||||
assert!(
|
||||
picker.handle_create_or_modify(base.join(&name)).is_some(),
|
||||
"round {round}: on_create_or_modify({name}) should succeed for create"
|
||||
);
|
||||
}
|
||||
|
||||
live_files.push(FileState {
|
||||
name,
|
||||
token,
|
||||
is_base: false,
|
||||
last_write_sec: epoch_secs(),
|
||||
});
|
||||
next_file_id += 1;
|
||||
op_counter += 1;
|
||||
} else if roll < 75 && !live_files.is_empty() {
|
||||
// ── DELETE existing file (15%) ──
|
||||
let idx = rng.next_u32() as usize % live_files.len();
|
||||
let removed = live_files.swap_remove(idx);
|
||||
let path = base.join(&removed.name);
|
||||
fs::remove_file(&path).unwrap();
|
||||
|
||||
{
|
||||
let mut guard = shared_picker.write().unwrap();
|
||||
let picker = guard.as_mut().unwrap();
|
||||
assert!(
|
||||
picker.remove_file_by_path(&path),
|
||||
"round {round}: remove_file_by_path({}) should succeed",
|
||||
removed.name
|
||||
);
|
||||
}
|
||||
|
||||
dead_tokens.push(removed.token);
|
||||
op_counter += 1;
|
||||
} else if roll < 85 && !live_files.is_empty() {
|
||||
// ── RENAME file (10%) ──
|
||||
let idx = rng.next_u32() as usize % live_files.len();
|
||||
let old_name = live_files[idx].name.clone();
|
||||
let old_path = base.join(&old_name);
|
||||
let content = fs::read_to_string(&old_path).unwrap();
|
||||
|
||||
// Remove old file from disk + picker.
|
||||
fs::remove_file(&old_path).unwrap();
|
||||
{
|
||||
let mut guard = shared_picker.write().unwrap();
|
||||
let picker = guard.as_mut().unwrap();
|
||||
picker.remove_file_by_path(&old_path);
|
||||
}
|
||||
|
||||
// Create new file with same content but different name.
|
||||
let new_name = format!("renamed_{next_file_id:04}.rs");
|
||||
fs::write(base.join(&new_name), &content).unwrap();
|
||||
{
|
||||
let mut guard = shared_picker.write().unwrap();
|
||||
let picker = guard.as_mut().unwrap();
|
||||
assert!(
|
||||
picker
|
||||
.handle_create_or_modify(base.join(&new_name))
|
||||
.is_some(),
|
||||
"round {round}: on_create_or_modify({new_name}) should succeed for rename"
|
||||
);
|
||||
}
|
||||
|
||||
live_files[idx].name = new_name;
|
||||
live_files[idx].is_base = false;
|
||||
live_files[idx].last_write_sec = epoch_secs();
|
||||
next_file_id += 1;
|
||||
op_counter += 1;
|
||||
}
|
||||
// else: no-op / read-only (15%) — just run verification below.
|
||||
|
||||
// ── VERIFY after every round ──
|
||||
{
|
||||
let guard = shared_picker.read().unwrap();
|
||||
let picker = guard.as_ref().unwrap();
|
||||
|
||||
for fs in &live_files {
|
||||
// Plain text grep: every live token must be found.
|
||||
let t0 = std::time::Instant::now();
|
||||
let plain_count = grep_plain_count(picker, &fs.token);
|
||||
t_grep_plain += t0.elapsed();
|
||||
grep_plain_calls.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
|
||||
assert!(
|
||||
plain_count >= 1,
|
||||
"round {round}: plain grep should find live token {} in {} (got {plain_count})",
|
||||
fs.token,
|
||||
fs.name
|
||||
);
|
||||
|
||||
// Regex grep: search with `{first5}.*{last5}` pattern.
|
||||
let regex_pattern = build_regex_pattern(&fs.token);
|
||||
let t0 = std::time::Instant::now();
|
||||
let regex_count = grep_regex_count(picker, ®ex_pattern);
|
||||
t_grep_regex += t0.elapsed();
|
||||
grep_regex_calls.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
|
||||
assert!(
|
||||
regex_count >= 1,
|
||||
"round {round}: regex grep '{}' should find live token {} in {} (got {regex_count})",
|
||||
regex_pattern,
|
||||
fs.token,
|
||||
fs.name
|
||||
);
|
||||
|
||||
// Fuzzy file search: every live file must be findable by name.
|
||||
let stem = extract_stem(&fs.name);
|
||||
let t0 = std::time::Instant::now();
|
||||
let fuzzy_results = fuzzy_search_paths(picker, &stem);
|
||||
t_fuzzy += t0.elapsed();
|
||||
fuzzy_calls.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
|
||||
assert!(
|
||||
fuzzy_results.iter().any(|p| p.contains(&fs.name)),
|
||||
"round {round}: fuzzy search '{}' should find file {} in results: {:?}",
|
||||
stem,
|
||||
fs.name,
|
||||
fuzzy_results
|
||||
);
|
||||
}
|
||||
|
||||
// Dead tokens must return 0 grep results.
|
||||
for dead in &dead_tokens {
|
||||
let t0 = std::time::Instant::now();
|
||||
let count = grep_plain_count(picker, dead);
|
||||
t_dead_check += t0.elapsed();
|
||||
dead_calls.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
|
||||
assert_eq!(
|
||||
count, 0,
|
||||
"round {round}: dead token {dead} should NOT be findable (got {count})"
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
let total = test_start.elapsed();
|
||||
let t_overhead = t_sleep + t_bigram_wait + t_git;
|
||||
let t_search = t_grep_plain + t_grep_regex + t_fuzzy + t_dead_check;
|
||||
let t_mutations = total.saturating_sub(t_overhead + t_search);
|
||||
let n_grep_plain = grep_plain_calls.load(std::sync::atomic::Ordering::Relaxed);
|
||||
let n_grep_regex = grep_regex_calls.load(std::sync::atomic::Ordering::Relaxed);
|
||||
let n_fuzzy = fuzzy_calls.load(std::sync::atomic::Ordering::Relaxed);
|
||||
let n_dead = dead_calls.load(std::sync::atomic::Ordering::Relaxed);
|
||||
eprintln!("\n╔══════════════════════════════════════════════════════╗");
|
||||
eprintln!("║ Fuzz Test Performance Breakdown ║");
|
||||
eprintln!("╠══════════════════════════════════════════════════════╣");
|
||||
eprintln!(
|
||||
"║ Total wall time: {:>8.1}ms ║",
|
||||
total.as_secs_f64() * 1000.0
|
||||
);
|
||||
eprintln!("║ ── Overhead ─────────────────────────────────────── ║");
|
||||
eprintln!(
|
||||
"║ Sleep (mtime waits): {:>8.1}ms ║",
|
||||
t_sleep.as_secs_f64() * 1000.0
|
||||
);
|
||||
eprintln!(
|
||||
"║ Git init+commit: {:>8.1}ms ║",
|
||||
t_git.as_secs_f64() * 1000.0
|
||||
);
|
||||
eprintln!(
|
||||
"║ Bigram index build+scan: {:>8.1}ms ║",
|
||||
t_bigram_wait.as_secs_f64() * 1000.0
|
||||
);
|
||||
eprintln!(
|
||||
"║ ── Search ({:>3} live files, {:>3} dead tokens) ────── ║",
|
||||
live_files.len(),
|
||||
dead_tokens.len()
|
||||
);
|
||||
eprintln!(
|
||||
"║ Plain grep: {:>4} calls {:>8.1}ms ({:>6.1}µs/call) ║",
|
||||
n_grep_plain,
|
||||
t_grep_plain.as_secs_f64() * 1000.0,
|
||||
t_grep_plain.as_secs_f64() * 1_000_000.0 / n_grep_plain.max(1) as f64
|
||||
);
|
||||
eprintln!(
|
||||
"║ Regex grep: {:>4} calls {:>8.1}ms ({:>6.1}µs/call) ║",
|
||||
n_grep_regex,
|
||||
t_grep_regex.as_secs_f64() * 1000.0,
|
||||
t_grep_regex.as_secs_f64() * 1_000_000.0 / n_grep_regex.max(1) as f64
|
||||
);
|
||||
eprintln!(
|
||||
"║ Fuzzy find: {:>4} calls {:>8.1}ms ({:>6.1}µs/call) ║",
|
||||
n_fuzzy,
|
||||
t_fuzzy.as_secs_f64() * 1000.0,
|
||||
t_fuzzy.as_secs_f64() * 1_000_000.0 / n_fuzzy.max(1) as f64
|
||||
);
|
||||
eprintln!(
|
||||
"║ Dead checks: {:>4} calls {:>8.1}ms ({:>6.1}µs/call) ║",
|
||||
n_dead,
|
||||
t_dead_check.as_secs_f64() * 1000.0,
|
||||
t_dead_check.as_secs_f64() * 1_000_000.0 / n_dead.max(1) as f64
|
||||
);
|
||||
eprintln!("║ ── Other ────────────────────────────────────────── ║");
|
||||
eprintln!(
|
||||
"║ Mutations + FS I/O: {:>8.1}ms ║",
|
||||
t_mutations.as_secs_f64() * 1000.0
|
||||
);
|
||||
eprintln!("╚══════════════════════════════════════════════════════╝");
|
||||
}
|
||||
|
||||
fn write_diverse_file(dir: &Path, name: &str, token: &str, index: usize) {
|
||||
let domain_idx = index % DOMAINS.len();
|
||||
write_diverse_file_with_domain(dir, name, token, domain_idx);
|
||||
}
|
||||
|
||||
fn write_diverse_file_with_domain(dir: &Path, name: &str, token: &str, domain_idx: usize) {
|
||||
let domain = DOMAINS[domain_idx % DOMAINS.len()];
|
||||
let content = format!(
|
||||
"// File: {name}\n\
|
||||
// Domain content for bigram diversity\n\
|
||||
{domain}\n\
|
||||
// === Unique searchable token below ===\n\
|
||||
const MARKER: &str = \"{token}\";\n\
|
||||
fn marker_function_{token}() {{ println!(\"{token}\"); }}\n"
|
||||
);
|
||||
|
||||
if let Some(parent) = PathBuf::from(name).parent() {
|
||||
if !parent.as_os_str().is_empty() {
|
||||
fs::create_dir_all(dir.join(parent)).unwrap();
|
||||
}
|
||||
}
|
||||
fs::write(dir.join(name), content).unwrap();
|
||||
}
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════════════
|
||||
// Search helpers
|
||||
// ═══════════════════════════════════════════════════════════════════════
|
||||
|
||||
fn grep_plain_opts() -> GrepSearchOptions {
|
||||
GrepSearchOptions {
|
||||
max_file_size: 10 * 1024 * 1024,
|
||||
max_matches_per_file: 200,
|
||||
smart_case: true,
|
||||
file_offset: 0,
|
||||
page_limit: 500,
|
||||
mode: GrepMode::PlainText,
|
||||
time_budget_ms: 0,
|
||||
before_context: 0,
|
||||
after_context: 0,
|
||||
classify_definitions: false,
|
||||
trim_whitespace: false,
|
||||
abort_signal: None,
|
||||
}
|
||||
}
|
||||
|
||||
fn grep_regex_opts() -> GrepSearchOptions {
|
||||
GrepSearchOptions {
|
||||
mode: GrepMode::Regex,
|
||||
..grep_plain_opts()
|
||||
}
|
||||
}
|
||||
|
||||
fn grep_plain_count(picker: &FilePicker, query: &str) -> usize {
|
||||
let parsed = parse_grep_query(query);
|
||||
picker.grep(&parsed, &grep_plain_opts()).matches.len()
|
||||
}
|
||||
|
||||
fn grep_regex_count(picker: &FilePicker, regex_query: &str) -> usize {
|
||||
let parsed = parse_grep_query(regex_query);
|
||||
picker.grep(&parsed, &grep_regex_opts()).matches.len()
|
||||
}
|
||||
|
||||
/// Build a regex pattern from a token: `{first5}.*{last5}`.
|
||||
/// For tokens shorter than 10 chars, just use the literal (escaped).
|
||||
fn build_regex_pattern(token: &str) -> String {
|
||||
if token.len() >= 10 {
|
||||
let first5 = &token[..5];
|
||||
let last5 = &token[token.len() - 5..];
|
||||
format!("{}.*{}", regex_escape(first5), regex_escape(last5))
|
||||
} else {
|
||||
regex_escape(token)
|
||||
}
|
||||
}
|
||||
|
||||
/// Escape regex metacharacters in a string.
|
||||
fn regex_escape(s: &str) -> String {
|
||||
let mut escaped = String::with_capacity(s.len() + 4);
|
||||
for ch in s.chars() {
|
||||
match ch {
|
||||
'.' | '*' | '+' | '?' | '(' | ')' | '[' | ']' | '{' | '}' | '\\' | '^' | '$' | '|' => {
|
||||
escaped.push('\\');
|
||||
escaped.push(ch);
|
||||
}
|
||||
_ => escaped.push(ch),
|
||||
}
|
||||
}
|
||||
escaped
|
||||
}
|
||||
|
||||
/// Extract a fuzzy-searchable stem from a filename.
|
||||
/// Strips the extension and any leading path components, keeping the bare name.
|
||||
fn extract_stem(name: &str) -> String {
|
||||
let p = PathBuf::from(name);
|
||||
p.file_stem()
|
||||
.unwrap_or_default()
|
||||
.to_string_lossy()
|
||||
.to_string()
|
||||
}
|
||||
|
||||
fn fuzzy_search_paths(picker: &FilePicker, query: &str) -> Vec<String> {
|
||||
let parser = QueryParser::default();
|
||||
let parsed = parser.parse(query);
|
||||
let result = picker.fuzzy_search(
|
||||
&parsed,
|
||||
None,
|
||||
FuzzySearchOptions {
|
||||
max_threads: 1,
|
||||
pagination: PaginationArgs {
|
||||
offset: 0,
|
||||
limit: 200,
|
||||
},
|
||||
..Default::default()
|
||||
},
|
||||
);
|
||||
result
|
||||
.items
|
||||
.iter()
|
||||
.map(|f| f.relative_path(picker))
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn wait_for_bigram(shared_picker: &SharedFilePicker) {
|
||||
let deadline = std::time::Instant::now() + Duration::from_secs(10);
|
||||
loop {
|
||||
std::thread::sleep(Duration::from_millis(50));
|
||||
let ready = shared_picker
|
||||
.read()
|
||||
.ok()
|
||||
.map(|guard| {
|
||||
guard
|
||||
.as_ref()
|
||||
.map_or(false, |p| !p.is_scan_active() && p.bigram_index().is_some())
|
||||
})
|
||||
.unwrap_or(false);
|
||||
if ready {
|
||||
break;
|
||||
}
|
||||
assert!(
|
||||
std::time::Instant::now() < deadline,
|
||||
"Timed out waiting for bigram build"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
fn git_run(dir: &Path, args: &[&str]) {
|
||||
let out = Command::new("git")
|
||||
.args(args)
|
||||
.current_dir(dir)
|
||||
.env("GIT_AUTHOR_NAME", "test")
|
||||
.env("GIT_AUTHOR_EMAIL", "test@test.com")
|
||||
.env("GIT_COMMITTER_NAME", "test")
|
||||
.env("GIT_COMMITTER_EMAIL", "test@test.com")
|
||||
.output()
|
||||
.unwrap_or_else(|e| panic!("git {:?} failed: {}", args, e));
|
||||
assert!(
|
||||
out.status.success(),
|
||||
"git {:?} failed: {}",
|
||||
args,
|
||||
String::from_utf8_lossy(&out.stderr)
|
||||
);
|
||||
}
|
||||
|
||||
fn epoch_secs() -> u64 {
|
||||
SystemTime::now()
|
||||
.duration_since(UNIX_EPOCH)
|
||||
.unwrap()
|
||||
.as_secs()
|
||||
}
|
||||
|
||||
fn git_init_and_commit(dir: &Path) {
|
||||
git_run(dir, &["init"]);
|
||||
git_run(dir, &["add", "-A"]);
|
||||
git_run(dir, &["commit", "-m", "initial"]);
|
||||
}
|
||||
|
||||
/// Proves that dropping the picker while post-scan (warmup + bigram build)
|
||||
/// is actively iterating raw pointers does NOT segfault. The Drop impl
|
||||
/// sets `cancelled`, waits for `post_scan_indexing_active` to clear, and
|
||||
/// only then frees the backing Vec.
|
||||
///
|
||||
/// Runs 10 iterations to exercise the race window reliably.
|
||||
#[test]
|
||||
fn drop_during_post_scan_does_not_crash() {
|
||||
let mut caught_active = 0u32;
|
||||
|
||||
for round in 0..10 {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let base = tmp.path();
|
||||
|
||||
// Create enough files so bigram build takes measurable time
|
||||
for i in 0..2000 {
|
||||
let dir = base.join(format!("d_{:02}", i % 20));
|
||||
fs::create_dir_all(&dir).unwrap();
|
||||
let content = format!(
|
||||
"fn func_{i}() {{ let x = {i}; println!(\"{{x}}\"); }}\n\
|
||||
const T_{i}: &str = \"TOKEN_{i}\";\n"
|
||||
);
|
||||
fs::write(dir.join(format!("f_{i:04}.rs")), content).unwrap();
|
||||
}
|
||||
|
||||
git_init_and_commit(base);
|
||||
|
||||
let shared_picker = SharedFilePicker::default();
|
||||
|
||||
FilePicker::new_with_shared_state(
|
||||
shared_picker.clone(),
|
||||
SharedFrecency::noop(),
|
||||
FilePickerOptions {
|
||||
base_path: base.to_string_lossy().to_string(),
|
||||
enable_mmap_cache: true,
|
||||
enable_content_indexing: true,
|
||||
watch: false,
|
||||
mode: FFFMode::Neovim,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
// Wait for scan but NOT for bigram — drop while post-scan is active
|
||||
shared_picker.wait_for_scan(Duration::from_secs(10));
|
||||
|
||||
// Poll until post_scan_indexing_active is true (bigram started)
|
||||
let deadline = std::time::Instant::now() + Duration::from_secs(5);
|
||||
let mut was_active = false;
|
||||
loop {
|
||||
if let Ok(guard) = shared_picker.read() {
|
||||
if let Some(picker) = guard.as_ref() {
|
||||
if picker.is_post_scan_active() {
|
||||
was_active = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
if std::time::Instant::now() > deadline {
|
||||
break;
|
||||
}
|
||||
std::thread::sleep(Duration::from_millis(1));
|
||||
}
|
||||
|
||||
if was_active {
|
||||
caught_active += 1;
|
||||
}
|
||||
|
||||
// Drop the picker while post_scan_indexing_active is set.
|
||||
// Take it out of the shared handle first, then drop outside the lock —
|
||||
// Drop spins until post-scan finishes, which needs the write lock for
|
||||
// bigram install, so we can't hold it during Drop.
|
||||
let old_picker = shared_picker.write().unwrap().take();
|
||||
drop(old_picker); // Drop fires here — spins until post-scan exits
|
||||
|
||||
assert!(
|
||||
shared_picker.read().unwrap().is_none(),
|
||||
"round {round}: picker should be None after drop"
|
||||
);
|
||||
}
|
||||
|
||||
// At least some rounds must have caught the post-scan active window
|
||||
assert!(
|
||||
caught_active > 0,
|
||||
"Test didn't catch post_scan_indexing_active=true in any round. \
|
||||
The test is not exercising the race. ({caught_active}/10)"
|
||||
);
|
||||
eprintln!("Caught post-scan active in {caught_active}/10 rounds");
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because one or more lines are too long
@@ -0,0 +1,740 @@
|
||||
//! Proptest-driven fuzz test against real GitHub repos with a live watcher.
|
||||
//!
|
||||
//! Clones real repository, runs the simulated close to real user sereies of file system ewvents and
|
||||
//! verifies that fff can still find the correct files. Test cases are randomized and preserved
|
||||
//! using proptest
|
||||
//!
|
||||
//! Run:
|
||||
//! ```sh
|
||||
//! RUSTFLAGS="--cfg stress" cargo test -p fff-search --test fuzz_real_repos -- --nocapture
|
||||
//! ```
|
||||
//!
|
||||
//! Increase coverage:
|
||||
//! ```sh
|
||||
//! FFF_FUZZ_CASES=4 FFF_FUZZ_MAX_OPS=60 \
|
||||
//! RUSTFLAGS="--cfg stress" cargo test -p fff-search --test fuzz_real_repos -- --nocapture
|
||||
//! ```
|
||||
#![cfg(stress)]
|
||||
|
||||
use std::fs;
|
||||
use std::path::{Path, PathBuf};
|
||||
use std::process::Command;
|
||||
use std::time::{Duration, Instant, SystemTime, UNIX_EPOCH};
|
||||
|
||||
use proptest::prelude::*;
|
||||
use proptest::test_runner::{Config as ProptestConfig, FileFailurePersistence};
|
||||
|
||||
use fff_search::file_picker::{FFFMode, FilePicker, is_known_binary_extension};
|
||||
use fff_search::grep::{GrepMode, GrepSearchOptions, parse_grep_query};
|
||||
use fff_search::{FilePickerOptions, SharedFilePicker, SharedFrecency};
|
||||
|
||||
const REPO_POOL: &[(&str, &str)] = &[
|
||||
("dmtrKovalenko/fff", "fff"),
|
||||
("BurntSushi/ripgrep", "ripgrep"),
|
||||
("sharkdp/fd", "fd"),
|
||||
("ogham/exa", "exa"),
|
||||
("casey/just", "just"),
|
||||
("ajeetdsouza/zoxide", "zoxide"),
|
||||
("helix-editor/helix", "helix"),
|
||||
("astral-sh/ruff", "ruff"),
|
||||
("biomejs/biome", "biome"),
|
||||
("denoland/deno_lint", "deno_lint"),
|
||||
("nickel-lang/nickel", "nickel"),
|
||||
("typst/typst", "typst"),
|
||||
("gleam-lang/gleam", "gleam"),
|
||||
("pretzelhammer/rust-blog", "rust-blog"),
|
||||
("tokio-rs/mini-redis", "mini-redis"),
|
||||
];
|
||||
|
||||
const CACHE_DIR: &str = "/tmp/fff_fuzz_repos";
|
||||
/// Fixed settle time for watcher event propagation.
|
||||
const WATCHER_SETTLE: Duration = Duration::from_millis(100);
|
||||
/// Maximum time to wait for watcher to process all pending events.
|
||||
const CONVERGE_TIMEOUT: Duration = Duration::from_secs(30);
|
||||
|
||||
fn fuzz_cases() -> u32 {
|
||||
std::env::var("FFF_FUZZ_CASES")
|
||||
.ok()
|
||||
.and_then(|v| v.parse().ok())
|
||||
.unwrap_or(2)
|
||||
}
|
||||
|
||||
fn fuzz_max_ops() -> usize {
|
||||
std::env::var("FFF_FUZZ_MAX_OPS")
|
||||
.ok()
|
||||
.and_then(|v| v.parse().ok())
|
||||
.unwrap_or(30)
|
||||
}
|
||||
|
||||
fn fuzz_min_ops() -> usize {
|
||||
std::env::var("FFF_FUZZ_MIN_OPS")
|
||||
.ok()
|
||||
.and_then(|v| v.parse().ok())
|
||||
.unwrap_or(15)
|
||||
}
|
||||
|
||||
fn ensure_repo_cloned(repo_url: &str, local_name: &str) -> PathBuf {
|
||||
let cache = PathBuf::from(CACHE_DIR);
|
||||
fs::create_dir_all(&cache).unwrap();
|
||||
let repo_path = cache.join(local_name);
|
||||
if repo_path.join(".git").exists() {
|
||||
return repo_path;
|
||||
}
|
||||
|
||||
let full_url = format!("https://github.com/{}.git", repo_url);
|
||||
eprintln!(" Cloning {} ...", full_url);
|
||||
let out = Command::new("git")
|
||||
.args(["clone", "--depth=1", "--single-branch", &full_url])
|
||||
.arg(&repo_path)
|
||||
.output()
|
||||
.expect("git clone failed");
|
||||
assert!(
|
||||
out.status.success(),
|
||||
"git clone {} failed: {}",
|
||||
full_url,
|
||||
String::from_utf8_lossy(&out.stderr)
|
||||
);
|
||||
repo_path
|
||||
}
|
||||
|
||||
fn copy_repo_to_workdir(cached: &Path, workdir: &Path) {
|
||||
let out = Command::new("cp")
|
||||
.args(["-r"])
|
||||
.arg(cached)
|
||||
.arg(workdir)
|
||||
.output()
|
||||
.expect("cp -r failed");
|
||||
assert!(
|
||||
out.status.success(),
|
||||
"cp -r failed: {}",
|
||||
String::from_utf8_lossy(&out.stderr)
|
||||
);
|
||||
}
|
||||
|
||||
fn collect_text_files(base: &Path) -> Vec<PathBuf> {
|
||||
// Use `git ls-files` without --cached to get only files that are both
|
||||
// tracked AND not gitignored. Files like Cargo.lock that are committed
|
||||
// but in .gitignore would appear with --cached but the fff picker skips
|
||||
// them during walk (respects .gitignore), causing false test failures.
|
||||
let out = Command::new("git")
|
||||
.args(["ls-files", "--others", "--exclude-standard", "-z"])
|
||||
.current_dir(base)
|
||||
.output()
|
||||
.unwrap();
|
||||
|
||||
// Get tracked files that aren't ignored
|
||||
let tracked = Command::new("git")
|
||||
.args(["ls-files", "-z"])
|
||||
.current_dir(base)
|
||||
.output()
|
||||
.unwrap();
|
||||
|
||||
// Check which tracked files are actually ignored
|
||||
let ignored_check = Command::new("git")
|
||||
.args(["check-ignore", "--stdin", "-z"])
|
||||
.stdin(std::process::Stdio::piped())
|
||||
.stdout(std::process::Stdio::piped())
|
||||
.current_dir(base)
|
||||
.spawn();
|
||||
|
||||
let mut ignored_set: std::collections::HashSet<String> = std::collections::HashSet::new();
|
||||
if let Ok(mut child) = ignored_check {
|
||||
use std::io::Write;
|
||||
if let Some(ref mut stdin) = child.stdin {
|
||||
let _ = stdin.write_all(&tracked.stdout);
|
||||
}
|
||||
if let Ok(output) = child.wait_with_output() {
|
||||
for path in output.stdout.split(|&b| b == 0) {
|
||||
if !path.is_empty() {
|
||||
if let Ok(s) = std::str::from_utf8(path) {
|
||||
ignored_set.insert(s.to_string());
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Combine: tracked non-ignored non-binary files
|
||||
let mut files: Vec<PathBuf> = Vec::new();
|
||||
for path in tracked.stdout.split(|&b| b == 0) {
|
||||
if path.is_empty() {
|
||||
continue;
|
||||
}
|
||||
let Ok(s) = std::str::from_utf8(path) else {
|
||||
continue;
|
||||
};
|
||||
if ignored_set.contains(s) {
|
||||
continue;
|
||||
}
|
||||
let full = base.join(s);
|
||||
if full.is_file() && !is_known_binary_extension(&full) {
|
||||
files.push(full);
|
||||
}
|
||||
}
|
||||
files
|
||||
}
|
||||
|
||||
/// Edit a file by injecting a marker line at a deterministic position,
|
||||
/// preserving the rest of the content. Returns the original line that was
|
||||
/// replaced so it can be restored on revert.
|
||||
fn inject_marker(path: &Path, marker: &str, seed: u32) -> Option<String> {
|
||||
let content = fs::read_to_string(path).ok()?;
|
||||
let lines: Vec<&str> = content.lines().collect();
|
||||
if lines.is_empty() {
|
||||
fs::write(path, format!("// {marker}\n")).ok()?;
|
||||
return Some(String::new());
|
||||
}
|
||||
|
||||
// Pick a stable line position based on seed and file length
|
||||
let line_idx = seed as usize % lines.len();
|
||||
let original_line = lines[line_idx].to_string();
|
||||
|
||||
let mut result = String::with_capacity(content.len() + marker.len() + 10);
|
||||
for (i, line) in lines.iter().enumerate() {
|
||||
if i == line_idx {
|
||||
result.push_str(&format!("// {marker}"));
|
||||
} else {
|
||||
result.push_str(line);
|
||||
}
|
||||
result.push('\n');
|
||||
}
|
||||
fs::write(path, &result).ok()?;
|
||||
Some(original_line)
|
||||
}
|
||||
|
||||
/// Revert a file by restoring the original line at the same position
|
||||
/// where inject_marker placed the marker.
|
||||
fn revert_marker(path: &Path, marker: &str, original_line: &str) {
|
||||
let Ok(content) = fs::read_to_string(path) else {
|
||||
return;
|
||||
};
|
||||
let marker_line = format!("// {marker}");
|
||||
let result: String = content
|
||||
.lines()
|
||||
.map(|l| if l == marker_line { original_line } else { l })
|
||||
.collect::<Vec<_>>()
|
||||
.join("\n")
|
||||
+ "\n";
|
||||
let _ = fs::write(path, result);
|
||||
}
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════════════════
|
||||
// Search helpers
|
||||
// ═══════════════════════════════════════════════════════════════════════════
|
||||
|
||||
fn grep_opts(mode: GrepMode) -> GrepSearchOptions {
|
||||
GrepSearchOptions {
|
||||
max_file_size: 10 * 1024 * 1024,
|
||||
max_matches_per_file: 200,
|
||||
smart_case: true,
|
||||
file_offset: 0,
|
||||
page_limit: 500,
|
||||
mode,
|
||||
time_budget_ms: 5000,
|
||||
before_context: 0,
|
||||
after_context: 0,
|
||||
classify_definitions: false,
|
||||
trim_whitespace: false,
|
||||
abort_signal: None,
|
||||
}
|
||||
}
|
||||
|
||||
fn grep_finds(picker: &FilePicker, query: &str, mode: GrepMode) -> bool {
|
||||
let parsed = parse_grep_query(query);
|
||||
let result = picker.grep(&parsed, &grep_opts(mode));
|
||||
!result.matches.is_empty()
|
||||
}
|
||||
|
||||
fn grep_file_list(picker: &FilePicker, query: &str, mode: GrepMode) -> Vec<String> {
|
||||
let parsed = parse_grep_query(query);
|
||||
let result = picker.grep(&parsed, &grep_opts(mode));
|
||||
result
|
||||
.files
|
||||
.iter()
|
||||
.map(|f| f.relative_path(picker))
|
||||
.collect()
|
||||
}
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════════════════
|
||||
// Infrastructure
|
||||
// ═══════════════════════════════════════════════════════════════════════════
|
||||
|
||||
fn wait_for_bigram(sp: &SharedFilePicker) {
|
||||
let deadline = Instant::now() + Duration::from_secs(120);
|
||||
loop {
|
||||
std::thread::sleep(Duration::from_millis(50));
|
||||
let ready = sp
|
||||
.read()
|
||||
.ok()
|
||||
.map(|g| {
|
||||
g.as_ref()
|
||||
.map_or(false, |p| !p.is_scan_active() && p.bigram_index().is_some())
|
||||
})
|
||||
.unwrap_or(false);
|
||||
if ready {
|
||||
return;
|
||||
}
|
||||
assert!(
|
||||
Instant::now() < deadline,
|
||||
"Timed out waiting for bigram index"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
fn epoch_secs() -> u64 {
|
||||
SystemTime::now()
|
||||
.duration_since(UNIX_EPOCH)
|
||||
.unwrap()
|
||||
.as_secs()
|
||||
}
|
||||
|
||||
struct TrackedFile {
|
||||
relative: String,
|
||||
marker: String,
|
||||
/// The original line content that was replaced, for revert
|
||||
original_line: String,
|
||||
is_created: bool,
|
||||
last_write_sec: u64,
|
||||
}
|
||||
|
||||
fn run_scenario(ops: &[Op]) {
|
||||
// Stream fff logs at info+ level by default. Override with RUST_LOG.
|
||||
let _ = tracing_subscriber::fmt()
|
||||
.with_env_filter(
|
||||
tracing_subscriber::EnvFilter::try_from_default_env()
|
||||
.unwrap_or_else(|_| tracing_subscriber::EnvFilter::new("warn,fff_search=info")),
|
||||
)
|
||||
.with_test_writer()
|
||||
.try_init();
|
||||
|
||||
// Allow forcing a specific repo via env for reproduction
|
||||
let repo_idx = std::env::var("FFF_FUZZ_REPO_IDX")
|
||||
.ok()
|
||||
.and_then(|v| v.parse::<usize>().ok())
|
||||
.unwrap_or_else(|| ops.len() % REPO_POOL.len());
|
||||
let (repo_url, local_name) = REPO_POOL[repo_idx];
|
||||
eprintln!("=== fuzz_real_repos: repo={repo_url} ops={} ===", ops.len());
|
||||
|
||||
let scenario_start = Instant::now();
|
||||
let cached = ensure_repo_cloned(repo_url, local_name);
|
||||
let tmp = tempfile::TempDir::new().unwrap();
|
||||
let workdir = tmp.path().join(local_name);
|
||||
copy_repo_to_workdir(&cached, &workdir);
|
||||
|
||||
// Ensure target/ is gitignored
|
||||
let gitignore = workdir.join(".gitignore");
|
||||
let mut gi = fs::read_to_string(&gitignore).unwrap_or_default();
|
||||
if !gi.contains("target/") {
|
||||
gi.push_str("\ntarget/\n");
|
||||
fs::write(&gitignore, &gi).unwrap();
|
||||
}
|
||||
|
||||
let shared_picker = SharedFilePicker::default();
|
||||
FilePicker::new_with_shared_state(
|
||||
shared_picker.clone(),
|
||||
SharedFrecency::noop(),
|
||||
FilePickerOptions {
|
||||
base_path: workdir.to_string_lossy().to_string(),
|
||||
enable_mmap_cache: true,
|
||||
enable_content_indexing: true,
|
||||
watch: true,
|
||||
mode: FFFMode::Neovim,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.expect("FilePicker init");
|
||||
|
||||
let t0 = Instant::now();
|
||||
wait_for_bigram(&shared_picker);
|
||||
let bigram_ms = t0.elapsed().as_secs_f64() * 1000.0;
|
||||
|
||||
{
|
||||
let guard = shared_picker.read().unwrap();
|
||||
let picker = guard.as_ref().unwrap();
|
||||
let file_count = picker.get_files().len();
|
||||
eprintln!(" indexed {file_count} files, bigram ready in {bigram_ms:.0}ms");
|
||||
}
|
||||
|
||||
// Advance mtime past scan timestamp
|
||||
std::thread::sleep(Duration::from_millis(1100));
|
||||
|
||||
let mut tracked: Vec<TrackedFile> = Vec::new();
|
||||
let mut dead_markers: Vec<String> = Vec::new();
|
||||
let mut ignored_markers: Vec<String> = Vec::new();
|
||||
let mut ops_since_verify: usize = 0;
|
||||
let mut text_files: Option<Vec<PathBuf>> = None;
|
||||
|
||||
for (op_idx, op) in ops.iter().enumerate() {
|
||||
match op {
|
||||
Op::CreateFile { seed } => {
|
||||
let name = format!("fff_fuzz_new_{seed:08x}.rs");
|
||||
let marker = format!("FFF_FUZZ_NEW_{seed:08x}");
|
||||
// Marker appears only once on its own line
|
||||
let content = format!("// {marker}\nfn placeholder() {{}}\n");
|
||||
fs::write(workdir.join(&name), content).unwrap();
|
||||
tracked.push(TrackedFile {
|
||||
relative: name,
|
||||
marker,
|
||||
original_line: String::new(),
|
||||
is_created: true,
|
||||
last_write_sec: epoch_secs(),
|
||||
});
|
||||
ops_since_verify += 1;
|
||||
}
|
||||
Op::EditTracked { seed } => {
|
||||
if tracked.is_empty() {
|
||||
continue;
|
||||
}
|
||||
let idx = *seed as usize % tracked.len();
|
||||
if tracked[idx].last_write_sec >= epoch_secs() {
|
||||
std::thread::sleep(Duration::from_millis(1100));
|
||||
}
|
||||
let new_marker = format!("FFF_FUZZ_EDIT_{seed:08x}");
|
||||
let path = workdir.join(&tracked[idx].relative);
|
||||
// Replace the line containing our old marker with the new one
|
||||
let old_marker_line = format!("// {}", tracked[idx].marker);
|
||||
let content = fs::read_to_string(&path).unwrap_or_default();
|
||||
let new_content = content
|
||||
.lines()
|
||||
.map(|l| {
|
||||
if l == old_marker_line {
|
||||
format!("// {new_marker}")
|
||||
} else {
|
||||
l.to_string()
|
||||
}
|
||||
})
|
||||
.collect::<Vec<_>>()
|
||||
.join("\n")
|
||||
+ "\n";
|
||||
fs::write(&path, new_content).unwrap();
|
||||
|
||||
dead_markers.push(tracked[idx].marker.clone());
|
||||
tracked[idx].marker = new_marker;
|
||||
tracked[idx].last_write_sec = epoch_secs();
|
||||
ops_since_verify += 1;
|
||||
}
|
||||
Op::EditRandom { seed } => {
|
||||
let files = text_files.get_or_insert_with(|| collect_text_files(&workdir));
|
||||
if files.is_empty() {
|
||||
continue;
|
||||
}
|
||||
let target = &files[*seed as usize % files.len()];
|
||||
let relative = target
|
||||
.strip_prefix(&workdir)
|
||||
.unwrap()
|
||||
.to_string_lossy()
|
||||
.to_string();
|
||||
|
||||
if let Some(t) = tracked.iter().find(|t| t.relative == relative) {
|
||||
if t.last_write_sec >= epoch_secs() {
|
||||
std::thread::sleep(Duration::from_millis(1100));
|
||||
}
|
||||
}
|
||||
|
||||
let marker = format!("FFF_FUZZ_RAND_{seed:08x}");
|
||||
|
||||
// If already tracked, replace old marker line
|
||||
if let Some(pos) = tracked.iter().position(|t| t.relative == relative) {
|
||||
let old_marker_line = format!("// {}", tracked[pos].marker);
|
||||
let content = fs::read_to_string(target).unwrap_or_default();
|
||||
let new_content = content
|
||||
.lines()
|
||||
.map(|l| {
|
||||
if l == old_marker_line {
|
||||
format!("// {marker}")
|
||||
} else {
|
||||
l.to_string()
|
||||
}
|
||||
})
|
||||
.collect::<Vec<_>>()
|
||||
.join("\n")
|
||||
+ "\n";
|
||||
fs::write(target, new_content).unwrap();
|
||||
dead_markers.push(tracked[pos].marker.clone());
|
||||
tracked[pos].marker = marker;
|
||||
tracked[pos].last_write_sec = epoch_secs();
|
||||
} else {
|
||||
// First edit: inject marker at a deterministic line
|
||||
let original = inject_marker(target, &marker, *seed).unwrap_or_default();
|
||||
tracked.push(TrackedFile {
|
||||
relative,
|
||||
marker,
|
||||
original_line: original,
|
||||
is_created: false,
|
||||
last_write_sec: epoch_secs(),
|
||||
});
|
||||
}
|
||||
ops_since_verify += 1;
|
||||
}
|
||||
Op::DeleteTracked => {
|
||||
if tracked.is_empty() {
|
||||
continue;
|
||||
}
|
||||
let removed = tracked.swap_remove(0);
|
||||
let abs = workdir.join(&removed.relative);
|
||||
if abs.exists() {
|
||||
if removed.is_created {
|
||||
fs::remove_file(&abs).ok();
|
||||
} else {
|
||||
let _ = Command::new("git")
|
||||
.args(["rm", "-f", &removed.relative])
|
||||
.current_dir(&workdir)
|
||||
.output();
|
||||
}
|
||||
}
|
||||
dead_markers.push(removed.marker);
|
||||
text_files = None; // invalidate cache after deletion
|
||||
ops_since_verify += 1;
|
||||
}
|
||||
Op::RevertTracked => {
|
||||
// Revert a non-created tracked file using `git checkout`
|
||||
// (restores original content, marker disappears)
|
||||
let revertable = tracked.iter().position(|t| !t.is_created);
|
||||
let Some(idx) = revertable else { continue };
|
||||
|
||||
if tracked[idx].last_write_sec >= epoch_secs() {
|
||||
std::thread::sleep(Duration::from_millis(1100));
|
||||
}
|
||||
|
||||
let _ = Command::new("git")
|
||||
.args(["checkout", "--", &tracked[idx].relative])
|
||||
.current_dir(&workdir)
|
||||
.output();
|
||||
|
||||
let reverted = tracked.swap_remove(idx);
|
||||
dead_markers.push(reverted.marker);
|
||||
text_files = None; // invalidate cache after revert
|
||||
ops_since_verify += 1;
|
||||
}
|
||||
Op::IgnoredBurst { count, seed } => {
|
||||
let dir = workdir.join("target/debug/build");
|
||||
fs::create_dir_all(&dir).unwrap();
|
||||
for i in 0..*count {
|
||||
let marker = format!("FFF_IGN_{seed:08x}_{i}");
|
||||
fs::write(
|
||||
dir.join(format!("ign_{seed:08x}_{i}.rs")),
|
||||
format!("// {marker}\nfn {marker}() {{}}\n"),
|
||||
)
|
||||
.unwrap();
|
||||
ignored_markers.push(marker);
|
||||
}
|
||||
ops_since_verify += 1;
|
||||
}
|
||||
Op::Verify => {
|
||||
if tracked.is_empty() && dead_markers.is_empty() {
|
||||
continue;
|
||||
}
|
||||
|
||||
// Poll until the watcher has propagated all pending events:
|
||||
// all live markers findable, all dead markers gone, no ignored leaks.
|
||||
let modes = [
|
||||
(GrepMode::PlainText, "Plain"),
|
||||
(GrepMode::Regex, "Regex"),
|
||||
(GrepMode::Fuzzy, "Fuzzy"),
|
||||
];
|
||||
let (mode, mode_name) = modes[op_idx % modes.len()];
|
||||
|
||||
let deadline = Instant::now() + CONVERGE_TIMEOUT;
|
||||
let mut last_failure: Option<String> = None;
|
||||
|
||||
loop {
|
||||
std::thread::sleep(WATCHER_SETTLE);
|
||||
|
||||
// Write trigger to force a watcher batch
|
||||
let trigger = workdir.join("fff_fuzz_trigger.rs");
|
||||
let _ = fs::write(&trigger, format!("// trigger {}\n", op_idx));
|
||||
|
||||
std::thread::sleep(WATCHER_SETTLE);
|
||||
|
||||
let mut all_ok = true;
|
||||
|
||||
// Check live markers (drop lock between each grep)
|
||||
for tf in &tracked {
|
||||
let guard = shared_picker.read().unwrap();
|
||||
let picker = guard.as_ref().unwrap();
|
||||
let found = grep_finds(picker, &tf.marker, mode);
|
||||
drop(guard);
|
||||
if !found {
|
||||
last_failure = Some(format!(
|
||||
"{mode_name} grep for {:?} in {:?} not found\n\
|
||||
is_created={} exists={} on_disk_has_marker={}",
|
||||
tf.marker,
|
||||
tf.relative,
|
||||
tf.is_created,
|
||||
workdir.join(&tf.relative).exists(),
|
||||
fs::read_to_string(workdir.join(&tf.relative))
|
||||
.map(|c| c.contains(&tf.marker))
|
||||
.unwrap_or(false),
|
||||
));
|
||||
all_ok = false;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
// Check dead markers (only sample a few per iteration to
|
||||
// avoid holding the lock too long with many dead markers)
|
||||
if all_ok {
|
||||
let sample_size = dead_markers.len().min(20);
|
||||
for dead in dead_markers.iter().take(sample_size) {
|
||||
let guard = shared_picker.read().unwrap();
|
||||
let picker = guard.as_ref().unwrap();
|
||||
let found = grep_finds(picker, dead, GrepMode::PlainText);
|
||||
drop(guard);
|
||||
if found {
|
||||
last_failure = Some(format!("dead marker {dead:?} still findable"));
|
||||
all_ok = false;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Check ignored markers (sample first 5)
|
||||
if all_ok {
|
||||
for ig in ignored_markers.iter().take(5) {
|
||||
let guard = shared_picker.read().unwrap();
|
||||
let picker = guard.as_ref().unwrap();
|
||||
let files = grep_file_list(picker, ig, GrepMode::PlainText);
|
||||
drop(guard);
|
||||
if !files.is_empty() {
|
||||
last_failure =
|
||||
Some(format!("ignored marker {ig:?} found in {files:?}"));
|
||||
all_ok = false;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if all_ok {
|
||||
eprintln!(
|
||||
" op[{op_idx}] verify OK: {mode_name} mode, {} live, {} dead, {} ignored",
|
||||
tracked.len(),
|
||||
dead_markers.len(),
|
||||
ignored_markers.len(),
|
||||
);
|
||||
break;
|
||||
}
|
||||
|
||||
if Instant::now() >= deadline {
|
||||
panic!(
|
||||
"op[{op_idx}] verify TIMEOUT after {CONVERGE_TIMEOUT:?}:\n {}\n ops_since_last_verify={}",
|
||||
last_failure.unwrap_or_default(),
|
||||
ops_since_verify,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
ops_since_verify = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Final convergence: poll until everything is consistent
|
||||
let deadline = Instant::now() + CONVERGE_TIMEOUT;
|
||||
loop {
|
||||
std::thread::sleep(WATCHER_SETTLE);
|
||||
let guard = shared_picker.read().unwrap();
|
||||
let picker = guard.as_ref().unwrap();
|
||||
|
||||
let live_ok = tracked
|
||||
.iter()
|
||||
.all(|tf| grep_finds(picker, &tf.marker, GrepMode::PlainText));
|
||||
let dead_ok = dead_markers
|
||||
.iter()
|
||||
.all(|d| !grep_finds(picker, d, GrepMode::PlainText));
|
||||
drop(guard);
|
||||
|
||||
if live_ok && dead_ok {
|
||||
break;
|
||||
}
|
||||
assert!(
|
||||
Instant::now() < deadline,
|
||||
"final verify TIMEOUT: live_ok={live_ok} dead_ok={dead_ok}"
|
||||
);
|
||||
}
|
||||
|
||||
// Teardown
|
||||
shared_picker.wait_for_indexing_complete(Duration::from_secs(30));
|
||||
if let Ok(mut guard) = shared_picker.write() {
|
||||
if let Some(mut picker) = guard.take() {
|
||||
picker.stop_background_monitor();
|
||||
}
|
||||
}
|
||||
|
||||
eprintln!(
|
||||
" PASSED: {} ops, {} tracked, {} dead, {} ignored ({:.1}s)",
|
||||
ops.len(),
|
||||
tracked.len(),
|
||||
dead_markers.len(),
|
||||
ignored_markers.len(),
|
||||
scenario_start.elapsed().as_secs_f64(),
|
||||
);
|
||||
}
|
||||
|
||||
// ================
|
||||
// Proptest harness
|
||||
// =================
|
||||
//
|
||||
fn proptest_config() -> ProptestConfig {
|
||||
ProptestConfig {
|
||||
cases: fuzz_cases(),
|
||||
max_shrink_iters: 0,
|
||||
fork: false,
|
||||
failure_persistence: Some(Box::new(FileFailurePersistence::Direct(concat!(
|
||||
env!("CARGO_MANIFEST_DIR"),
|
||||
"/tests/fuzz_real_repos.proptest-regressions",
|
||||
)))),
|
||||
..ProptestConfig::default()
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone)]
|
||||
enum Op {
|
||||
/// Create a new file with a unique marker
|
||||
CreateFile { seed: u32 },
|
||||
/// Edit a tracked file, replacing the marker line with a new marker
|
||||
EditTracked { seed: u32 },
|
||||
/// Edit a random repo file, injecting a marker at a deterministic line
|
||||
EditRandom { seed: u32 },
|
||||
/// Delete a tracked file
|
||||
DeleteTracked,
|
||||
/// Revert a tracked edit, restoring the original line (marker disappears)
|
||||
RevertTracked,
|
||||
/// Burst of writes into ignored directory
|
||||
IgnoredBurst { count: u8, seed: u32 },
|
||||
/// Search verification round (no mutation)
|
||||
Verify,
|
||||
}
|
||||
|
||||
fn op_strategy() -> impl Strategy<Value = Op> {
|
||||
prop_oneof![
|
||||
// Create new files — exercises overflow path
|
||||
12 => any::<u32>().prop_map(|s| Op::CreateFile { seed: s }),
|
||||
// Edit tracked files — exercises content invalidation
|
||||
18 => any::<u32>().prop_map(|s| Op::EditTracked { seed: s }),
|
||||
// Edit random repo files — exercises bigram overlay for base files
|
||||
18 => any::<u32>().prop_map(|s| Op::EditRandom { seed: s }),
|
||||
// Delete tracked files — exercises tombstoning
|
||||
8 => Just(Op::DeleteTracked),
|
||||
// Revert tracked edits — marker must disappear from search
|
||||
10 => Just(Op::RevertTracked),
|
||||
// Burst ignored writes — exercises .gitignore filtering under load
|
||||
9 => (1u8..20, any::<u32>()).prop_map(|(c, s)| Op::IgnoredBurst { count: c, seed: s }),
|
||||
// Explicit verification rounds
|
||||
25 => Just(Op::Verify),
|
||||
]
|
||||
}
|
||||
|
||||
fn ops_strategy() -> impl Strategy<Value = Vec<Op>> {
|
||||
let min = fuzz_min_ops();
|
||||
let max = fuzz_max_ops();
|
||||
prop::collection::vec(op_strategy(), min..=max)
|
||||
}
|
||||
|
||||
proptest! {
|
||||
#![proptest_config(proptest_config())]
|
||||
|
||||
#[test]
|
||||
fn fuzz_real_repos_proptest(ops in ops_strategy()) {
|
||||
run_scenario(&ops);
|
||||
}
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,455 @@
|
||||
//! Reproduces the deadlock/hang caused by LMDB writer mutex contention.
|
||||
//!
|
||||
//! When another process holds the LMDB writer mutex (via a long-running write
|
||||
//! transaction or because it crashed without releasing it), any call to
|
||||
//! `write_txn()` blocks indefinitely — including on the neovim main thread
|
||||
//! during `QueryTracker::open()` or frecency `track_access()`.
|
||||
//!
|
||||
//! In production this manifests as neovim hanging on startup:
|
||||
//! require('fff.core').ensure_initialized()
|
||||
//! → init_db() → QueryTracker::open() → write_txn() → HANGS
|
||||
//!
|
||||
//! Or during normal use when BufEnter fires:
|
||||
//! track_access → frecency.track_access() → write_txn() → HANGS
|
||||
//!
|
||||
//! Reproduction: fork a child process that holds the LMDB write lock
|
||||
//! indefinitely, then attempt to use the same database from the parent.
|
||||
//! The parent's `write_txn()` blocks on the cross-process writer mutex.
|
||||
//!
|
||||
//! This test confirms that the current code has NO timeout or fallback when the
|
||||
//! LMDB writer mutex is unavailable — making it vulnerable to indefinite hangs
|
||||
//! whenever another process (fff-mcp, another neovim, or a crashed instance)
|
||||
//! holds or has stuck the mutex.
|
||||
|
||||
#![cfg(unix)]
|
||||
|
||||
use std::fs;
|
||||
use std::path::{Path, PathBuf};
|
||||
use std::sync::mpsc;
|
||||
use std::time::Duration;
|
||||
|
||||
use fff_search::frecency::FrecencyTracker;
|
||||
use fff_search::query_tracker::QueryTracker;
|
||||
|
||||
/// Returns whether `f` completes within `timeout`.
|
||||
fn completes_within(
|
||||
label: &'static str,
|
||||
timeout: Duration,
|
||||
f: impl FnOnce() + Send + 'static,
|
||||
) -> bool {
|
||||
let (tx, rx) = mpsc::channel::<()>();
|
||||
let _worker = std::thread::Builder::new()
|
||||
.name(format!("deadlock-repro-{label}"))
|
||||
.spawn(move || {
|
||||
f();
|
||||
let _ = tx.send(());
|
||||
})
|
||||
.expect("spawn worker");
|
||||
|
||||
rx.recv_timeout(timeout).is_ok()
|
||||
}
|
||||
|
||||
/// Fork a child that opens the LMDB env and holds a write transaction
|
||||
/// indefinitely (simulating a stuck/long-running process). Returns the
|
||||
/// child PID so the parent can kill it during cleanup.
|
||||
fn fork_child_holding_write_lock(db_path: &Path) -> libc::pid_t {
|
||||
let db_path_str = db_path.to_str().unwrap().to_owned();
|
||||
|
||||
let mut pipe_fds: [libc::c_int; 2] = [0; 2];
|
||||
assert_eq!(unsafe { libc::pipe(pipe_fds.as_mut_ptr()) }, 0);
|
||||
let read_fd = pipe_fds[0];
|
||||
let write_fd = pipe_fds[1];
|
||||
|
||||
let child_pid = unsafe { libc::fork() };
|
||||
match child_pid {
|
||||
-1 => panic!("fork() failed: {}", std::io::Error::last_os_error()),
|
||||
0 => {
|
||||
// === CHILD PROCESS ===
|
||||
unsafe { libc::close(read_fd) };
|
||||
|
||||
let env = unsafe {
|
||||
let mut opts = heed::EnvOpenOptions::new();
|
||||
opts.map_size(10 * 1024 * 1024);
|
||||
opts.open(Path::new(&db_path_str)).expect("child: open env")
|
||||
};
|
||||
|
||||
// Acquire the cross-process writer mutex via write_txn
|
||||
let _wtxn = env.write_txn().expect("child: write_txn");
|
||||
|
||||
// Signal parent that the lock is held
|
||||
unsafe { libc::write(write_fd, b"R".as_ptr() as *const libc::c_void, 1) };
|
||||
|
||||
// Hold the lock forever — parent will eventually kill us
|
||||
loop {
|
||||
unsafe { libc::pause() };
|
||||
}
|
||||
}
|
||||
pid => {
|
||||
// === PARENT PROCESS ===
|
||||
unsafe { libc::close(write_fd) };
|
||||
|
||||
// Wait for child to confirm it holds the write lock
|
||||
let mut buf = [0u8; 1];
|
||||
let n = unsafe { libc::read(read_fd, buf.as_mut_ptr() as *mut libc::c_void, 1) };
|
||||
assert_eq!(n, 1, "child didn't signal readiness");
|
||||
assert_eq!(buf[0], b'R');
|
||||
unsafe { libc::close(read_fd) };
|
||||
|
||||
pid
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Kill and reap the child process.
|
||||
fn kill_child(pid: libc::pid_t) {
|
||||
unsafe {
|
||||
libc::kill(pid, libc::SIGKILL);
|
||||
let mut status: libc::c_int = 0;
|
||||
libc::waitpid(pid, &mut status, 0);
|
||||
}
|
||||
}
|
||||
|
||||
/// Verify QueryTracker works correctly after close+reopen — the
|
||||
/// open_database_safe path must find existing named databases via read txn.
|
||||
#[test]
|
||||
fn lmdb_reopen_finds_existing_databases() {
|
||||
let tmp = tempfile::TempDir::new().unwrap();
|
||||
let db_path = tmp.path().join("lmdb_reopen");
|
||||
fs::create_dir_all(&db_path).unwrap();
|
||||
|
||||
// First open: creates the databases via write_txn fallback
|
||||
{
|
||||
let mut tracker = QueryTracker::open(&db_path).unwrap();
|
||||
let project = Path::new("/test/project");
|
||||
let file = Path::new("/test/project/src/main.rs");
|
||||
tracker
|
||||
.track_query_completion("hello", project, file)
|
||||
.unwrap();
|
||||
}
|
||||
|
||||
// Second open: must find existing databases via read txn (no write_txn needed)
|
||||
{
|
||||
let tracker = QueryTracker::open(&db_path).unwrap();
|
||||
let project = Path::new("/test/project");
|
||||
let result = tracker.get_historical_query(project, 0).unwrap();
|
||||
assert_eq!(
|
||||
result,
|
||||
Some("hello".to_string()),
|
||||
"Query history should persist across close/reopen"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/// Env var the test binary checks on startup. When set, the binary skips the
|
||||
/// test harness and runs as a child worker instead. This avoids fork() in a
|
||||
/// multi-threaded parent — which copies mutex/allocator state from threads
|
||||
/// that no longer exist in the child and can deadlock heed/libc.
|
||||
const CHILD_MODE_ENV: &str = "FFF_PARALLEL_OPEN_CLOSE_CHILD";
|
||||
|
||||
/// Runs before the test harness when `CHILD_MODE_ENV` is set. Re-exec of
|
||||
/// the test binary lets us start child workers without forking from a
|
||||
/// multi-threaded parent.
|
||||
#[ctor::ctor]
|
||||
fn maybe_enter_child_mode() {
|
||||
if let Ok(spec) = std::env::var(CHILD_MODE_ENV) {
|
||||
let code = run_child_from_spec(&spec);
|
||||
std::process::exit(code);
|
||||
}
|
||||
}
|
||||
|
||||
/// Spec format: `db_path|idx|iterations|writer(0|1)`
|
||||
fn run_child_from_spec(spec: &str) -> i32 {
|
||||
let parts: Vec<&str> = spec.split('|').collect();
|
||||
if parts.len() != 4 {
|
||||
return CHILD_BAD_SPEC;
|
||||
}
|
||||
let db_path = parts[0];
|
||||
let idx: usize = match parts[1].parse() {
|
||||
Ok(v) => v,
|
||||
Err(_) => return CHILD_BAD_SPEC,
|
||||
};
|
||||
let iterations: usize = match parts[2].parse() {
|
||||
Ok(v) => v,
|
||||
Err(_) => return CHILD_BAD_SPEC,
|
||||
};
|
||||
let is_writer = parts[3] == "1";
|
||||
child_open_close_loop(db_path, iterations, idx, is_writer)
|
||||
}
|
||||
|
||||
const CHILD_OK: i32 = 0;
|
||||
const CHILD_OPEN_FAILED: i32 = 10;
|
||||
const CHILD_READ_FAILED: i32 = 11;
|
||||
const CHILD_WRITE_FAILED: i32 = 12;
|
||||
const CHILD_BAD_SPEC: i32 = 13;
|
||||
|
||||
fn child_open_close_loop(db_path: &str, iterations: usize, idx: usize, is_writer: bool) -> i32 {
|
||||
let project = Path::new("/test/project");
|
||||
for i in 0..iterations {
|
||||
let tracker = match QueryTracker::open(Path::new(db_path)) {
|
||||
Ok(t) => t,
|
||||
Err(e) => {
|
||||
eprintln!("child {idx} iter {i} reader open failed: {e:?}");
|
||||
return CHILD_OPEN_FAILED;
|
||||
}
|
||||
};
|
||||
if let Err(e) = tracker.get_historical_query(project, 0) {
|
||||
eprintln!("child {idx} iter {i} read failed: {e:?}");
|
||||
return CHILD_READ_FAILED;
|
||||
}
|
||||
drop(tracker);
|
||||
|
||||
if is_writer {
|
||||
let mut tracker = match QueryTracker::open(Path::new(db_path)) {
|
||||
Ok(t) => t,
|
||||
Err(e) => {
|
||||
eprintln!("child {idx} iter {i} writer open failed: {e:?}");
|
||||
return CHILD_OPEN_FAILED;
|
||||
}
|
||||
};
|
||||
let file = PathBuf::from(format!("/test/project/c{idx}_{i}.rs"));
|
||||
if let Err(e) = tracker.track_query_completion(&format!("q{idx}_{i}"), project, &file) {
|
||||
eprintln!("child {idx} iter {i} write failed: {e:?}");
|
||||
return CHILD_WRITE_FAILED;
|
||||
}
|
||||
drop(tracker);
|
||||
}
|
||||
}
|
||||
CHILD_OK
|
||||
}
|
||||
|
||||
/// Spawn `n` child processes via `Command::new(current_exe)`. No fork, so
|
||||
/// mutex/allocator state is not inherited. `writers` children also issue
|
||||
/// writes; the rest only read.
|
||||
fn spawn_open_close_children(
|
||||
db_path: &Path,
|
||||
n: usize,
|
||||
writers: usize,
|
||||
ops_per_child: usize,
|
||||
) -> Vec<std::process::Child> {
|
||||
assert!(writers <= n);
|
||||
let exe = std::env::current_exe().expect("current_exe");
|
||||
let db_path_str = db_path.to_str().unwrap().to_owned();
|
||||
|
||||
(0..n)
|
||||
.map(|idx| {
|
||||
let is_writer = idx < writers;
|
||||
let spec = format!(
|
||||
"{db_path_str}|{idx}|{ops_per_child}|{}",
|
||||
if is_writer { 1 } else { 0 }
|
||||
);
|
||||
std::process::Command::new(&exe)
|
||||
.env(CHILD_MODE_ENV, spec)
|
||||
.env_remove("RUST_LOG")
|
||||
.stdin(std::process::Stdio::null())
|
||||
.stdout(std::process::Stdio::null())
|
||||
.stderr(std::process::Stdio::inherit())
|
||||
.spawn()
|
||||
.expect("spawn child")
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Wait for every child with a per-call deadline. On timeout, kill and reap
|
||||
/// remaining children and return an Err describing the stuck set.
|
||||
fn wait_all_with_deadline(
|
||||
mut children: Vec<std::process::Child>,
|
||||
deadline: std::time::Instant,
|
||||
) -> Result<(), String> {
|
||||
let mut failures: Vec<(u32, Option<i32>)> = Vec::new();
|
||||
let mut remaining: Vec<std::process::Child> = Vec::new();
|
||||
|
||||
for mut child in children.drain(..) {
|
||||
loop {
|
||||
match child.try_wait() {
|
||||
Ok(Some(status)) => {
|
||||
let code = status.code();
|
||||
if code != Some(CHILD_OK) {
|
||||
failures.push((child.id(), code));
|
||||
}
|
||||
break;
|
||||
}
|
||||
Ok(None) => {
|
||||
if std::time::Instant::now() >= deadline {
|
||||
remaining.push(child);
|
||||
break;
|
||||
}
|
||||
std::thread::sleep(Duration::from_millis(25));
|
||||
}
|
||||
Err(e) => {
|
||||
failures.push((child.id(), None));
|
||||
let _ = e;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if !remaining.is_empty() {
|
||||
let stuck: Vec<u32> = remaining.iter().map(|c| c.id()).collect();
|
||||
for child in &mut remaining {
|
||||
let _ = child.kill();
|
||||
let _ = child.wait();
|
||||
}
|
||||
return Err(format!(
|
||||
"deadline exceeded; children still running: {stuck:?}"
|
||||
));
|
||||
}
|
||||
|
||||
if !failures.is_empty() {
|
||||
return Err(format!("children failed: {failures:?}"));
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Many processes open/close `QueryTracker` against the same DB path.
|
||||
/// Readers only: seeds once, then spawns N reader children.
|
||||
///
|
||||
/// heed 0.22 forbids opening the same env twice *within* one process
|
||||
/// (EnvAlreadyOpened), so cross-process contention is the right axis.
|
||||
#[test]
|
||||
fn query_tracker_many_parallel_open_close_same_path_readers() {
|
||||
let tmp = tempfile::TempDir::new().unwrap();
|
||||
let db_path = tmp.path().join("parallel_open_close_readers");
|
||||
fs::create_dir_all(&db_path).unwrap();
|
||||
|
||||
{
|
||||
let mut tracker = QueryTracker::open(&db_path).unwrap();
|
||||
tracker
|
||||
.track_query_completion(
|
||||
"seed",
|
||||
Path::new("/test/project"),
|
||||
Path::new("/test/project/src/main.rs"),
|
||||
)
|
||||
.unwrap();
|
||||
}
|
||||
|
||||
const N: usize = 8;
|
||||
const OPS: usize = 4;
|
||||
|
||||
let children = spawn_open_close_children(&db_path, N, 0, OPS);
|
||||
let deadline = std::time::Instant::now() + Duration::from_secs(30);
|
||||
wait_all_with_deadline(children, deadline).expect("parallel open/close (readers)");
|
||||
|
||||
let tracker = QueryTracker::open(&db_path).expect("post-storm reopen");
|
||||
let project = Path::new("/test/project");
|
||||
let result = tracker.get_historical_query(project, 0).unwrap();
|
||||
assert_eq!(
|
||||
result,
|
||||
Some("seed".to_string()),
|
||||
"Seed query should still be readable after parallel open/close storm"
|
||||
);
|
||||
}
|
||||
|
||||
/// Stronger variant: multiple processes race opens that both read AND write.
|
||||
/// LMDB serializes writers via a cross-process mutex; test that serialization
|
||||
/// makes forward progress and open/close pairs don't deadlock.
|
||||
#[test]
|
||||
fn query_tracker_parallel_open_write_close_same_path() {
|
||||
let tmp = tempfile::TempDir::new().unwrap();
|
||||
let db_path = tmp.path().join("parallel_open_write_close");
|
||||
fs::create_dir_all(&db_path).unwrap();
|
||||
|
||||
{
|
||||
let mut tracker = QueryTracker::open(&db_path).unwrap();
|
||||
tracker
|
||||
.track_query_completion(
|
||||
"seed",
|
||||
Path::new("/test/project"),
|
||||
Path::new("/test/project/src/main.rs"),
|
||||
)
|
||||
.unwrap();
|
||||
}
|
||||
|
||||
const N: usize = 4;
|
||||
const WRITERS: usize = 4;
|
||||
const OPS: usize = 3;
|
||||
|
||||
let children = spawn_open_close_children(&db_path, N, WRITERS, OPS);
|
||||
let deadline = std::time::Instant::now() + Duration::from_secs(30);
|
||||
wait_all_with_deadline(children, deadline).expect("parallel open/write/close");
|
||||
|
||||
let tracker = QueryTracker::open(&db_path).expect("post-storm reopen");
|
||||
let project = Path::new("/test/project");
|
||||
let seed = tracker.get_historical_query(project, 0).unwrap();
|
||||
assert!(
|
||||
seed.is_some(),
|
||||
"Env unreadable after parallel open/write/close storm"
|
||||
);
|
||||
}
|
||||
|
||||
/// Within a single process, opening the same env path twice concurrently is
|
||||
/// forbidden by heed — but a strict sequential open→use→drop→open loop must
|
||||
/// succeed every iteration. Regression guard for the reopen path.
|
||||
#[test]
|
||||
fn query_tracker_sequential_reopen_loop_same_path() {
|
||||
let tmp = tempfile::TempDir::new().unwrap();
|
||||
let db_path = tmp.path().join("sequential_reopen_loop");
|
||||
fs::create_dir_all(&db_path).unwrap();
|
||||
|
||||
{
|
||||
let mut tracker = QueryTracker::open(&db_path).unwrap();
|
||||
tracker
|
||||
.track_query_completion(
|
||||
"seed",
|
||||
Path::new("/test/project"),
|
||||
Path::new("/test/project/src/main.rs"),
|
||||
)
|
||||
.unwrap();
|
||||
}
|
||||
|
||||
for i in 0..64 {
|
||||
let mut tracker = QueryTracker::open(&db_path).expect("sequential reopen");
|
||||
let project = Path::new("/test/project");
|
||||
let file = PathBuf::from(format!("/test/project/iter_{i}.rs"));
|
||||
tracker
|
||||
.track_query_completion(&format!("iter_{i}"), project, &file)
|
||||
.expect("sequential track");
|
||||
drop(tracker);
|
||||
}
|
||||
|
||||
let tracker = QueryTracker::open(&db_path).expect("final reopen");
|
||||
let project = Path::new("/test/project");
|
||||
assert!(tracker.get_historical_query(project, 0).unwrap().is_some());
|
||||
}
|
||||
|
||||
/// When the frecency DB doesn't exist yet, `FrecencyTracker::open()` falls
|
||||
/// through to `write_txn()` + `create_database()`. This blocks if another
|
||||
/// process holds the writer mutex. This is the first-launch path.
|
||||
///
|
||||
/// NOTE: this test is disabled because heed 0.22 appears to use a
|
||||
/// try-then-create pattern for unnamed databases that doesn't always block.
|
||||
/// The QueryTracker test above (named databases, always needs write_txn)
|
||||
/// reliably demonstrates the same underlying issue.
|
||||
#[test]
|
||||
#[ignore = "heed 0.22 unnamed db creation may not require writer mutex in all cases"]
|
||||
fn frecency_open_blocks_on_fresh_db_when_another_process_holds_write_lock() {
|
||||
let tmp = tempfile::TempDir::new().unwrap();
|
||||
let db_path = tmp.path().join("frecency_fresh_deadlock");
|
||||
fs::create_dir_all(&db_path).unwrap();
|
||||
|
||||
let env = unsafe {
|
||||
let mut opts = heed::EnvOpenOptions::new();
|
||||
opts.map_size(10 * 1024 * 1024);
|
||||
opts.open(&db_path).unwrap()
|
||||
};
|
||||
drop(env);
|
||||
|
||||
let child_pid = fork_child_holding_write_lock(&db_path);
|
||||
|
||||
let db_path_clone = db_path.clone();
|
||||
let completed = completes_within(
|
||||
"FrecencyTracker::open (fresh db) while writer held",
|
||||
Duration::from_secs(3),
|
||||
move || {
|
||||
let _result = FrecencyTracker::open(&db_path_clone);
|
||||
},
|
||||
);
|
||||
|
||||
kill_child(child_pid);
|
||||
|
||||
assert!(
|
||||
!completed,
|
||||
"Expected FrecencyTracker::open() on a fresh DB to block (writer mutex \
|
||||
held by another process), but it completed."
|
||||
);
|
||||
}
|
||||
@@ -0,0 +1,539 @@
|
||||
//! Integration test: verifying that the background watcher dynamically detects
|
||||
//! newly created directories and picks up files written inside them.
|
||||
//!
|
||||
//! This covers the NonRecursive watching behavior where:
|
||||
//! 1. The watcher starts with watches on directories discovered during the
|
||||
//! initial scan.
|
||||
//! 2. A brand-new subdirectory is created at runtime (after the scan).
|
||||
//! 3. The watcher's event handler detects the directory Create event,
|
||||
//! collects it, and sends it to the owner thread via `watch_tx`.
|
||||
//! 4. The owner thread adds a NonRecursive watch on the new directory and
|
||||
//! does a flat (non-recursive) read_dir to inject files that already
|
||||
//! exist (race-window coverage).
|
||||
//! 5. Files created *after* the watch is established are picked up via
|
||||
//! normal event delivery.
|
||||
//!
|
||||
//! The test uses the real `BackgroundWatcher` (via `watch: true`) and polls
|
||||
//! the picker until the expected files appear or a timeout expires.
|
||||
|
||||
use std::fs;
|
||||
use std::path::Path;
|
||||
use std::process::Command;
|
||||
use std::time::{Duration, Instant};
|
||||
use tempfile::TempDir;
|
||||
|
||||
use fff_search::file_picker::{FFFMode, FilePicker};
|
||||
use fff_search::grep::{GrepMode, GrepSearchOptions, parse_grep_query};
|
||||
use fff_search::{
|
||||
FilePickerOptions, PaginationArgs, QueryParser, SharedFilePicker, SharedFrecency,
|
||||
};
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════════════
|
||||
// Helpers
|
||||
// ═══════════════════════════════════════════════════════════════════════
|
||||
|
||||
fn git_run(dir: &Path, args: &[&str]) {
|
||||
let out = Command::new("git")
|
||||
.args(args)
|
||||
.current_dir(dir)
|
||||
.env("GIT_AUTHOR_NAME", "test")
|
||||
.env("GIT_AUTHOR_EMAIL", "test@test.com")
|
||||
.env("GIT_COMMITTER_NAME", "test")
|
||||
.env("GIT_COMMITTER_EMAIL", "test@test.com")
|
||||
.output()
|
||||
.unwrap_or_else(|e| panic!("git {:?} failed: {}", args, e));
|
||||
assert!(
|
||||
out.status.success(),
|
||||
"git {:?} failed: {}",
|
||||
args,
|
||||
String::from_utf8_lossy(&out.stderr)
|
||||
);
|
||||
}
|
||||
|
||||
fn git_init_and_commit(dir: &Path) {
|
||||
git_run(dir, &["init", "-b", "main"]);
|
||||
git_run(dir, &["add", "-A"]);
|
||||
git_run(dir, &["commit", "-m", "initial"]);
|
||||
}
|
||||
|
||||
fn make_watched_picker(base: &Path) -> (SharedFilePicker, SharedFrecency) {
|
||||
let shared_picker = SharedFilePicker::default();
|
||||
let shared_frecency = SharedFrecency::noop();
|
||||
|
||||
FilePicker::new_with_shared_state(
|
||||
shared_picker.clone(),
|
||||
shared_frecency.clone(),
|
||||
FilePickerOptions {
|
||||
base_path: base.to_string_lossy().to_string(),
|
||||
enable_mmap_cache: false,
|
||||
mode: FFFMode::Neovim,
|
||||
watch: true,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.expect("Failed to create FilePicker");
|
||||
|
||||
(shared_picker, shared_frecency)
|
||||
}
|
||||
|
||||
/// Wait for the initial scan + watcher to be fully ready.
|
||||
fn wait_ready(shared_picker: &SharedFilePicker) {
|
||||
assert!(
|
||||
shared_picker.wait_for_scan(Duration::from_secs(10)),
|
||||
"Timed out waiting for initial scan"
|
||||
);
|
||||
assert!(
|
||||
shared_picker.wait_for_watcher(Duration::from_secs(10)),
|
||||
"Timed out waiting for watcher"
|
||||
);
|
||||
}
|
||||
|
||||
/// Poll the picker until `predicate` returns true or timeout expires.
|
||||
/// Returns the elapsed duration if successful, panics on timeout.
|
||||
fn poll_until(
|
||||
shared_picker: &SharedFilePicker,
|
||||
timeout: Duration,
|
||||
description: &str,
|
||||
predicate: impl Fn(&FilePicker) -> bool,
|
||||
) -> Duration {
|
||||
let start = Instant::now();
|
||||
loop {
|
||||
{
|
||||
let guard = shared_picker.read().unwrap();
|
||||
if let Some(ref picker) = *guard {
|
||||
if predicate(picker) {
|
||||
return start.elapsed();
|
||||
}
|
||||
}
|
||||
}
|
||||
if start.elapsed() >= timeout {
|
||||
// One final attempt to give a useful error message.
|
||||
let guard = shared_picker.read().unwrap();
|
||||
let picker = guard.as_ref().unwrap();
|
||||
let file_count = picker.get_files().len();
|
||||
let paths: Vec<String> = picker
|
||||
.get_files()
|
||||
.iter()
|
||||
.map(|f| f.relative_path(picker))
|
||||
.collect();
|
||||
panic!(
|
||||
"Timed out after {:?} waiting for: {}\n\
|
||||
Current file count: {}\n\
|
||||
Current files: {:?}",
|
||||
timeout, description, file_count, paths
|
||||
);
|
||||
}
|
||||
std::thread::sleep(Duration::from_millis(50));
|
||||
}
|
||||
}
|
||||
|
||||
fn grep_plain_count(picker: &FilePicker, query: &str) -> usize {
|
||||
let parsed = parse_grep_query(query);
|
||||
let opts = GrepSearchOptions {
|
||||
max_file_size: 10 * 1024 * 1024,
|
||||
max_matches_per_file: 200,
|
||||
smart_case: true,
|
||||
file_offset: 0,
|
||||
page_limit: 500,
|
||||
mode: GrepMode::PlainText,
|
||||
time_budget_ms: 0,
|
||||
before_context: 0,
|
||||
after_context: 0,
|
||||
classify_definitions: false,
|
||||
trim_whitespace: false,
|
||||
abort_signal: None,
|
||||
};
|
||||
picker.grep(&parsed, &opts).matches.len()
|
||||
}
|
||||
|
||||
fn fuzzy_search_paths(picker: &FilePicker, query: &str) -> Vec<String> {
|
||||
let parser = QueryParser::default();
|
||||
let parsed = parser.parse(query);
|
||||
let result = picker.fuzzy_search(
|
||||
&parsed,
|
||||
None,
|
||||
fff_search::FuzzySearchOptions {
|
||||
max_threads: 1,
|
||||
pagination: PaginationArgs {
|
||||
offset: 0,
|
||||
limit: 200,
|
||||
},
|
||||
..Default::default()
|
||||
},
|
||||
);
|
||||
result
|
||||
.items
|
||||
.iter()
|
||||
.map(|f| f.relative_path(picker))
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Debounce timeout in the watcher is 250ms. Events need to propagate through
|
||||
/// the debouncer, the owner thread park loop (1s), and the picker write lock.
|
||||
/// We use a generous timeout for CI environments.
|
||||
const WATCHER_TIMEOUT: Duration = Duration::from_secs(10);
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════════════
|
||||
// Tests
|
||||
// ═══════════════════════════════════════════════════════════════════════
|
||||
|
||||
/// Create a new directory and immediately write a file inside it.
|
||||
/// The file is written before the watch is registered, so the flat
|
||||
/// inject_existing_files scan in the owner thread must catch it.
|
||||
#[test]
|
||||
fn new_directory_and_file_detected_by_watcher() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let base = tmp.path().canonicalize().unwrap();
|
||||
|
||||
// Seed the repo with some initial files so the scan has something.
|
||||
fs::create_dir_all(base.join("src")).unwrap();
|
||||
fs::write(
|
||||
base.join("src/main.rs"),
|
||||
"fn main() { println!(\"INITIAL_MARKER\"); }\n",
|
||||
)
|
||||
.unwrap();
|
||||
fs::write(base.join("README.md"), "# Test project\n").unwrap();
|
||||
|
||||
git_init_and_commit(&base);
|
||||
|
||||
let (shared_picker, _frecency) = make_watched_picker(&base);
|
||||
wait_ready(&shared_picker);
|
||||
|
||||
// Sanity: initial file is indexed.
|
||||
poll_until(
|
||||
&shared_picker,
|
||||
Duration::from_secs(5),
|
||||
"initial file src/main.rs indexed",
|
||||
|picker| {
|
||||
picker
|
||||
.get_files()
|
||||
.iter()
|
||||
.any(|f| f.relative_path(picker).contains("main.rs"))
|
||||
},
|
||||
);
|
||||
|
||||
// Create a new directory and write a file into it immediately.
|
||||
// The file exists before the watch is registered — inject_existing_files
|
||||
// in the owner thread catches it via a flat read_dir.
|
||||
let new_dir = base.join("src/components");
|
||||
fs::create_dir_all(&new_dir).unwrap();
|
||||
fs::write(
|
||||
new_dir.join("button.rs"),
|
||||
"pub struct Button;\nconst TOKEN: &str = \"NEW_DIR_BUTTON_TOKEN\";\n",
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
// Wait for the watcher to detect the new directory + file.
|
||||
let elapsed = poll_until(
|
||||
&shared_picker,
|
||||
WATCHER_TIMEOUT,
|
||||
"file src/components/button.rs in new directory",
|
||||
|picker| {
|
||||
picker
|
||||
.get_files()
|
||||
.iter()
|
||||
.any(|f| f.relative_path(picker).contains("button.rs"))
|
||||
},
|
||||
);
|
||||
eprintln!(
|
||||
" New directory + file detected in {:.0}ms",
|
||||
elapsed.as_secs_f64() * 1000.0
|
||||
);
|
||||
|
||||
// Also verify via grep that the content is accessible.
|
||||
poll_until(
|
||||
&shared_picker,
|
||||
WATCHER_TIMEOUT,
|
||||
"grep finds NEW_DIR_BUTTON_TOKEN",
|
||||
|picker| grep_plain_count(picker, "NEW_DIR_BUTTON_TOKEN") >= 1,
|
||||
);
|
||||
|
||||
// And via fuzzy search.
|
||||
poll_until(
|
||||
&shared_picker,
|
||||
WATCHER_TIMEOUT,
|
||||
"fuzzy search finds button.rs",
|
||||
|picker| {
|
||||
let results = fuzzy_search_paths(picker, "button");
|
||||
results.iter().any(|p| p.contains("button.rs"))
|
||||
},
|
||||
);
|
||||
}
|
||||
|
||||
/// Create a new directory, then create files AFTER a delay to ensure the
|
||||
/// watch was established on the directory.
|
||||
#[test]
|
||||
fn file_created_after_directory_watch_established() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let base = tmp.path().canonicalize().unwrap();
|
||||
|
||||
fs::create_dir_all(base.join("lib")).unwrap();
|
||||
fs::write(base.join("lib/utils.rs"), "pub fn helper() {}\n").unwrap();
|
||||
|
||||
git_init_and_commit(&base);
|
||||
|
||||
let (shared_picker, _frecency) = make_watched_picker(&base);
|
||||
wait_ready(&shared_picker);
|
||||
|
||||
// Create the directory first, wait for the watcher to register it.
|
||||
let new_dir = base.join("lib/models");
|
||||
fs::create_dir(&new_dir).unwrap();
|
||||
|
||||
// Wait long enough for the debouncer to flush + owner thread to add watch.
|
||||
std::thread::sleep(Duration::from_millis(2000));
|
||||
|
||||
// Now write a file into the already-watched directory.
|
||||
fs::write(
|
||||
new_dir.join("user.rs"),
|
||||
"pub struct User { name: String }\nconst TOKEN: &str = \"POST_WATCH_USER_TOKEN\";\n",
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
let elapsed = poll_until(
|
||||
&shared_picker,
|
||||
WATCHER_TIMEOUT,
|
||||
"file lib/models/user.rs created after directory watch",
|
||||
|picker| {
|
||||
picker
|
||||
.get_files()
|
||||
.iter()
|
||||
.any(|f| f.relative_path(picker).contains("user.rs"))
|
||||
},
|
||||
);
|
||||
eprintln!(
|
||||
" Post-watch file detected in {:.0}ms",
|
||||
elapsed.as_secs_f64() * 1000.0
|
||||
);
|
||||
|
||||
// Grep sanity.
|
||||
poll_until(
|
||||
&shared_picker,
|
||||
WATCHER_TIMEOUT,
|
||||
"grep finds POST_WATCH_USER_TOKEN",
|
||||
|picker| grep_plain_count(picker, "POST_WATCH_USER_TOKEN") >= 1,
|
||||
);
|
||||
}
|
||||
|
||||
/// Create a deeply nested directory tree all at once with create_dir_all
|
||||
/// and write a file at the leaf. The watcher must detect the top-level
|
||||
/// directory via the parent's watch, inject_existing_files finds the file
|
||||
/// at the leaf (and intermediate dirs get their own watches from Create
|
||||
/// events on subsequent levels).
|
||||
#[test]
|
||||
fn deeply_nested_new_directories_detected() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let base = tmp.path().canonicalize().unwrap();
|
||||
|
||||
fs::write(base.join("root.txt"), "root file\n").unwrap();
|
||||
|
||||
git_init_and_commit(&base);
|
||||
|
||||
let (shared_picker, _frecency) = make_watched_picker(&base);
|
||||
wait_ready(&shared_picker);
|
||||
|
||||
// Create each level one at a time, waiting for each watch to register.
|
||||
// inject_existing_files is flat (non-recursive), so deeply nested dirs
|
||||
// need each parent to be watched before we can see files at the leaf.
|
||||
fs::create_dir(base.join("app")).unwrap();
|
||||
std::thread::sleep(Duration::from_millis(2000));
|
||||
|
||||
fs::create_dir(base.join("app/services")).unwrap();
|
||||
std::thread::sleep(Duration::from_millis(2000));
|
||||
|
||||
fs::create_dir(base.join("app/services/auth")).unwrap();
|
||||
// Write the file immediately — inject_existing_files catches it.
|
||||
fs::write(
|
||||
base.join("app/services/auth/jwt.rs"),
|
||||
"pub fn verify_token() {}\nconst TOKEN: &str = \"DEEP_NESTED_JWT_TOKEN\";\n",
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
let elapsed = poll_until(
|
||||
&shared_picker,
|
||||
WATCHER_TIMEOUT,
|
||||
"deeply nested file app/services/auth/jwt.rs",
|
||||
|picker| {
|
||||
picker
|
||||
.get_files()
|
||||
.iter()
|
||||
.any(|f| f.relative_path(picker).contains("jwt.rs"))
|
||||
},
|
||||
);
|
||||
eprintln!(
|
||||
" Deeply nested file detected in {:.0}ms",
|
||||
elapsed.as_secs_f64() * 1000.0
|
||||
);
|
||||
|
||||
// Verify content is grepable.
|
||||
poll_until(
|
||||
&shared_picker,
|
||||
WATCHER_TIMEOUT,
|
||||
"grep finds DEEP_NESTED_JWT_TOKEN",
|
||||
|picker| grep_plain_count(picker, "DEEP_NESTED_JWT_TOKEN") >= 1,
|
||||
);
|
||||
|
||||
// Now create a sibling at the same depth — the parent (app/services)
|
||||
// is already watched, so this just needs the flat inject.
|
||||
let sibling_dir = base.join("app/services/database");
|
||||
fs::create_dir(&sibling_dir).unwrap();
|
||||
fs::write(
|
||||
sibling_dir.join("pool.rs"),
|
||||
"pub struct ConnectionPool;\nconst TOKEN: &str = \"SIBLING_POOL_TOKEN\";\n",
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
let elapsed = poll_until(
|
||||
&shared_picker,
|
||||
WATCHER_TIMEOUT,
|
||||
"sibling nested file app/services/database/pool.rs",
|
||||
|picker| {
|
||||
picker
|
||||
.get_files()
|
||||
.iter()
|
||||
.any(|f| f.relative_path(picker).contains("pool.rs"))
|
||||
},
|
||||
);
|
||||
eprintln!(
|
||||
" Sibling nested file detected in {:.0}ms",
|
||||
elapsed.as_secs_f64() * 1000.0
|
||||
);
|
||||
|
||||
poll_until(
|
||||
&shared_picker,
|
||||
WATCHER_TIMEOUT,
|
||||
"grep finds SIBLING_POOL_TOKEN",
|
||||
|picker| grep_plain_count(picker, "SIBLING_POOL_TOKEN") >= 1,
|
||||
);
|
||||
}
|
||||
|
||||
/// Create a new directory and immediately burst-write multiple files.
|
||||
/// inject_existing_files catches all of them in one flat read_dir.
|
||||
#[test]
|
||||
fn burst_file_creation_in_new_directory() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let base = tmp.path().canonicalize().unwrap();
|
||||
|
||||
fs::create_dir_all(base.join("src")).unwrap();
|
||||
fs::write(base.join("src/lib.rs"), "// lib\n").unwrap();
|
||||
|
||||
git_init_and_commit(&base);
|
||||
|
||||
let (shared_picker, _frecency) = make_watched_picker(&base);
|
||||
wait_ready(&shared_picker);
|
||||
|
||||
// Create a new directory and immediately write 5 files.
|
||||
let batch_dir = base.join("src/batch");
|
||||
fs::create_dir(&batch_dir).unwrap();
|
||||
|
||||
let file_count = 5;
|
||||
for i in 0..file_count {
|
||||
fs::write(
|
||||
batch_dir.join(format!("item_{i}.rs")),
|
||||
format!("pub struct Item{i};\nconst TOKEN: &str = \"BATCH_ITEM_{i}\";\n"),
|
||||
)
|
||||
.unwrap();
|
||||
}
|
||||
|
||||
// Wait for ALL files to appear.
|
||||
let elapsed = poll_until(
|
||||
&shared_picker,
|
||||
WATCHER_TIMEOUT,
|
||||
&format!("all {file_count} batch files in src/batch/"),
|
||||
|picker| {
|
||||
let batch_count = picker
|
||||
.get_files()
|
||||
.iter()
|
||||
.filter(|f| {
|
||||
let p = f.relative_path(picker);
|
||||
p.starts_with("src/batch/") || p.starts_with("src\\batch\\")
|
||||
})
|
||||
.count();
|
||||
batch_count >= file_count
|
||||
},
|
||||
);
|
||||
eprintln!(
|
||||
" All {} burst files detected in {:.0}ms",
|
||||
file_count,
|
||||
elapsed.as_secs_f64() * 1000.0
|
||||
);
|
||||
|
||||
// Verify each file's content is grepable.
|
||||
for i in 0..file_count {
|
||||
let token = format!("BATCH_ITEM_{i}");
|
||||
poll_until(
|
||||
&shared_picker,
|
||||
WATCHER_TIMEOUT,
|
||||
&format!("grep finds {token}"),
|
||||
|picker| grep_plain_count(picker, &token) >= 1,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/// Verify that gitignored directories created at runtime are NOT watched
|
||||
/// and their files do NOT appear in the index.
|
||||
#[test]
|
||||
fn gitignored_new_directory_excluded() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let base = tmp.path().canonicalize().unwrap();
|
||||
|
||||
fs::write(base.join("main.rs"), "fn main() {}\n").unwrap();
|
||||
// Ignore the build/ directory.
|
||||
fs::write(base.join(".gitignore"), "build/\n").unwrap();
|
||||
|
||||
git_init_and_commit(&base);
|
||||
|
||||
let (shared_picker, _frecency) = make_watched_picker(&base);
|
||||
wait_ready(&shared_picker);
|
||||
|
||||
// Create a gitignored directory with files.
|
||||
let ignored_dir = base.join("build");
|
||||
fs::create_dir(&ignored_dir).unwrap();
|
||||
fs::write(
|
||||
ignored_dir.join("output.rs"),
|
||||
"const TOKEN: &str = \"IGNORED_BUILD_TOKEN\";\n",
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
// Also create a non-ignored directory to confirm the watcher works.
|
||||
let good_dir = base.join("src");
|
||||
fs::create_dir(&good_dir).unwrap();
|
||||
fs::write(
|
||||
good_dir.join("app.rs"),
|
||||
"const TOKEN: &str = \"GOOD_SRC_TOKEN\";\n",
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
// Wait for the non-ignored file to appear (proves watcher is working).
|
||||
poll_until(
|
||||
&shared_picker,
|
||||
WATCHER_TIMEOUT,
|
||||
"non-ignored file src/app.rs appears",
|
||||
|picker| {
|
||||
picker
|
||||
.get_files()
|
||||
.iter()
|
||||
.any(|f| f.relative_path(picker).contains("app.rs"))
|
||||
},
|
||||
);
|
||||
|
||||
// Give extra time for any straggler events from the ignored dir.
|
||||
std::thread::sleep(Duration::from_secs(2));
|
||||
|
||||
// The gitignored file must NOT be in the index.
|
||||
{
|
||||
let guard = shared_picker.read().unwrap();
|
||||
let picker = guard.as_ref().unwrap();
|
||||
|
||||
let has_ignored = picker
|
||||
.get_files()
|
||||
.iter()
|
||||
.any(|f| f.relative_path(picker).contains("output.rs"));
|
||||
assert!(
|
||||
!has_ignored,
|
||||
"Gitignored file build/output.rs should NOT be in the index"
|
||||
);
|
||||
|
||||
let grep_count = grep_plain_count(picker, "IGNORED_BUILD_TOKEN");
|
||||
assert_eq!(grep_count, 0, "Gitignored content should NOT be grepable");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,248 @@
|
||||
//! Regression test for https://github.com/dmtrKovalenko/fff/issues/381
|
||||
//!
|
||||
//! Directory (`PathSegment`) and file-path (`FilePath`) constraints must
|
||||
//! return results on every platform. Indexed paths on Windows use native
|
||||
//! backslash separators, so constraint matching has to accept either `/`
|
||||
//! or `\\` as a path boundary.
|
||||
|
||||
use std::fs;
|
||||
use std::path::Path;
|
||||
use tempfile::TempDir;
|
||||
|
||||
use fff_search::file_picker::FilePicker;
|
||||
use fff_search::grep::{GrepMode, GrepSearchOptions, parse_grep_query};
|
||||
use fff_search::{Constraint, FilePickerOptions, FuzzySearchOptions, PaginationArgs, QueryParser};
|
||||
|
||||
fn create_picker(base: &Path, specs: &[(&str, &str)]) -> FilePicker {
|
||||
for (rel, contents) in specs {
|
||||
let full_path = base.join(rel);
|
||||
if let Some(parent) = full_path.parent() {
|
||||
fs::create_dir_all(parent).unwrap();
|
||||
}
|
||||
fs::write(&full_path, contents).unwrap();
|
||||
}
|
||||
let mut picker = FilePicker::new(FilePickerOptions {
|
||||
base_path: base.to_string_lossy().to_string(),
|
||||
enable_mmap_cache: false,
|
||||
watch: false,
|
||||
..Default::default()
|
||||
})
|
||||
.expect("failed to create FilePicker");
|
||||
picker.collect_files().expect("failed to collect files");
|
||||
picker
|
||||
}
|
||||
|
||||
fn plain_opts() -> GrepSearchOptions {
|
||||
GrepSearchOptions {
|
||||
max_file_size: 10 * 1024 * 1024,
|
||||
max_matches_per_file: 200,
|
||||
smart_case: true,
|
||||
file_offset: 0,
|
||||
page_limit: 200,
|
||||
mode: GrepMode::PlainText,
|
||||
time_budget_ms: 0,
|
||||
before_context: 0,
|
||||
after_context: 0,
|
||||
classify_definitions: false,
|
||||
trim_whitespace: false,
|
||||
abort_signal: None,
|
||||
}
|
||||
}
|
||||
|
||||
fn fuzzy_search_paths(picker: &FilePicker, query: &str) -> Vec<String> {
|
||||
let parser = QueryParser::default();
|
||||
let parsed = parser.parse(query);
|
||||
let result = picker.fuzzy_search(
|
||||
&parsed,
|
||||
None,
|
||||
FuzzySearchOptions {
|
||||
max_threads: 1,
|
||||
pagination: PaginationArgs {
|
||||
offset: 0,
|
||||
limit: 200,
|
||||
},
|
||||
..Default::default()
|
||||
},
|
||||
);
|
||||
result
|
||||
.items
|
||||
.iter()
|
||||
.map(|f| f.relative_path(picker))
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Treat a relative path as a sequence of components regardless of the
|
||||
/// native separator so assertions are portable across Linux, macOS, Windows.
|
||||
fn has_segment(path: &str, segment: &str) -> bool {
|
||||
path.split(['/', '\\']).any(|s| s == segment)
|
||||
}
|
||||
|
||||
/// `grep handleRequest src/` — PathSegment constraint must match a nested
|
||||
/// `src` directory on every platform.
|
||||
#[test]
|
||||
fn grep_with_path_segment_constraint_nested() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let picker = create_picker(
|
||||
tmp.path(),
|
||||
&[
|
||||
("app/modules/src/services/handler.lua", "handleRequest()\n"),
|
||||
("app/modules/lib/util.lua", "handleRequest()\n"),
|
||||
("src/main.rs", "fn handleRequest() {}\n"),
|
||||
],
|
||||
);
|
||||
|
||||
let parsed = parse_grep_query("handleRequest src/");
|
||||
let result = picker.grep(&parsed, &plain_opts());
|
||||
|
||||
let matched_paths: Vec<String> = result
|
||||
.files
|
||||
.iter()
|
||||
.map(|f| f.relative_path(&picker))
|
||||
.collect();
|
||||
assert_eq!(
|
||||
result.matches.len(),
|
||||
2,
|
||||
"expected matches in two src/ files, got {matched_paths:?}"
|
||||
);
|
||||
for p in &matched_paths {
|
||||
assert!(
|
||||
has_segment(p, "src"),
|
||||
"every matched file must live under a `src` segment, got {p:?}"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/// `multi_grep` with a `PathSegment` constraint.
|
||||
#[test]
|
||||
fn multi_grep_with_path_segment_constraint() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let picker = create_picker(
|
||||
tmp.path(),
|
||||
&[
|
||||
(
|
||||
"app/modules/src/controller.lua",
|
||||
"handleRequest\nprocessJob\n",
|
||||
),
|
||||
("app/modules/lib/helper.lua", "handleRequest\n"),
|
||||
("app/src/legacy.lua", "processJob\n"),
|
||||
],
|
||||
);
|
||||
|
||||
let constraints = [Constraint::PathSegment("src")];
|
||||
let patterns = ["handleRequest", "processJob"];
|
||||
let result = picker.multi_grep(&patterns, &constraints, &plain_opts());
|
||||
|
||||
assert!(
|
||||
!result.matches.is_empty(),
|
||||
"multi_grep with `src/` constraint should return matches"
|
||||
);
|
||||
|
||||
let matched_paths: Vec<String> = result
|
||||
.files
|
||||
.iter()
|
||||
.map(|f| f.relative_path(&picker))
|
||||
.collect();
|
||||
for p in &matched_paths {
|
||||
assert!(
|
||||
has_segment(p, "src"),
|
||||
"every matched file must live under a `src` segment, got {p:?}"
|
||||
);
|
||||
}
|
||||
assert!(matched_paths.iter().any(|p| p.contains("controller.lua")));
|
||||
assert!(matched_paths.iter().any(|p| p.contains("legacy.lua")));
|
||||
}
|
||||
|
||||
/// Fuzzy search (`find_files src/ Controller`) must apply the path-segment
|
||||
/// filter to paths stored during indexing.
|
||||
#[test]
|
||||
fn fuzzy_search_with_path_segment_constraint() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let picker = create_picker(
|
||||
tmp.path(),
|
||||
&[
|
||||
("app/modules/src/services/BaseController.lua", "base\n"),
|
||||
("app/modules/src/services/UserController.lua", "user\n"),
|
||||
("app/modules/lib/BaseController.lua", "lib base\n"),
|
||||
("tests/src/MockController.lua", "mock\n"),
|
||||
],
|
||||
);
|
||||
|
||||
let results = fuzzy_search_paths(&picker, "src/ Controller");
|
||||
|
||||
assert!(
|
||||
!results.is_empty(),
|
||||
"fuzzy search with `src/` constraint should return results"
|
||||
);
|
||||
for p in &results {
|
||||
assert!(
|
||||
has_segment(p, "src"),
|
||||
"every result must live under `src`, got {p:?}"
|
||||
);
|
||||
}
|
||||
assert!(results.iter().any(|p| p.contains("BaseController")));
|
||||
assert!(results.iter().any(|p| p.contains("UserController")));
|
||||
assert!(results.iter().any(|p| p.contains("MockController")));
|
||||
}
|
||||
|
||||
/// `FilePath` suffix constraint must match stored paths even when components
|
||||
/// are separated by the platform-native separator during indexing.
|
||||
#[test]
|
||||
fn multi_grep_with_file_path_suffix_constraint() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let picker = create_picker(
|
||||
tmp.path(),
|
||||
&[
|
||||
("app/modules/src/services/handler.lua", "handleRequest\n"),
|
||||
("other/src/services/handler.lua", "handleRequest\n"),
|
||||
("app/modules/src/services/other.lua", "handleRequest\n"),
|
||||
],
|
||||
);
|
||||
|
||||
let constraints = [Constraint::FilePath("services/handler.lua")];
|
||||
let patterns = ["handleRequest"];
|
||||
let result = picker.multi_grep(&patterns, &constraints, &plain_opts());
|
||||
|
||||
let paths: Vec<String> = result
|
||||
.files
|
||||
.iter()
|
||||
.map(|f| f.relative_path(&picker))
|
||||
.collect();
|
||||
assert_eq!(
|
||||
paths.len(),
|
||||
2,
|
||||
"expected two matches for services/handler.lua, got {paths:?}"
|
||||
);
|
||||
for p in &paths {
|
||||
let ends_with_services_handler =
|
||||
p.ends_with("services/handler.lua") || p.ends_with("services\\handler.lua");
|
||||
assert!(
|
||||
ends_with_services_handler,
|
||||
"matched path must end with services/handler.lua, got {p:?}"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/// Glob constraints must match native Windows paths — the picker normalises
|
||||
/// separators when handing paths to the glob matcher.
|
||||
#[test]
|
||||
fn fuzzy_search_with_glob_constraint_matches_on_windows_paths() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let picker = create_picker(
|
||||
tmp.path(),
|
||||
&[
|
||||
("app/src/components/Button.lua", "\n"),
|
||||
("app/src/services/handler.lua", "\n"),
|
||||
("app/lib/components/Ignored.lua", "\n"),
|
||||
],
|
||||
);
|
||||
|
||||
let results = fuzzy_search_paths(&picker, "**/src/**/*.lua");
|
||||
assert!(
|
||||
results.iter().any(|p| p.contains("Button.lua")),
|
||||
"glob `**/src/**/*.lua` must match files below any `src/`, got {results:?}"
|
||||
);
|
||||
assert!(
|
||||
results.iter().any(|p| p.contains("handler.lua")),
|
||||
"glob `**/src/**/*.lua` must match services/handler.lua, got {results:?}"
|
||||
);
|
||||
}
|
||||
@@ -0,0 +1,165 @@
|
||||
//! Regression test: stopping the background watcher while the caller
|
||||
//! holds the [`SharedFilePicker`] write lock must NOT deadlock.
|
||||
//!
|
||||
//! There are two lock-ordering hazards the watcher has to handle:
|
||||
//!
|
||||
//! 1. The debouncer's event thread calls our handler, which wants
|
||||
//! `shared_picker.write()` to apply events. `stop()` used to
|
||||
//! `join()` that thread under the caller's write guard.
|
||||
//!
|
||||
//! 2. The owner thread registers new-directory watches and injects
|
||||
//! their existing files. Previously it held the debouncer mutex
|
||||
//! across `shared_picker.write()`, while `stop()` takes the
|
||||
//! debouncer mutex under the caller's write guard — inverse
|
||||
//! lock orders, classic deadlock.
|
||||
//!
|
||||
//! macOS FSEvents is the reliable reproducer for (1) because fresh
|
||||
//! `fs::write()` calls inside a just-watched temp dir queue events
|
||||
//! faster than the debounce tick can drain them. Creating new
|
||||
//! subdirectories exercises (2) via the owner thread's `watch_tx`.
|
||||
|
||||
use std::fs;
|
||||
use std::sync::mpsc;
|
||||
use std::time::Duration;
|
||||
use tempfile::TempDir;
|
||||
|
||||
use fff_search::file_picker::{FFFMode, FilePicker};
|
||||
use fff_search::{FilePickerOptions, SharedFilePicker, SharedFrecency};
|
||||
|
||||
/// Run `f` on a worker thread, require it to finish within `timeout`,
|
||||
/// panic with `msg` otherwise. The caller gets to describe what the
|
||||
/// worker is doing so a hung test produces an actionable message.
|
||||
fn run_with_deadlock_guard(
|
||||
msg: &'static str,
|
||||
timeout: Duration,
|
||||
f: impl FnOnce() + Send + 'static,
|
||||
) {
|
||||
let (done_tx, done_rx) = mpsc::channel::<()>();
|
||||
let worker = std::thread::Builder::new()
|
||||
.name("deadlock-guard-worker".into())
|
||||
.spawn(move || {
|
||||
f();
|
||||
let _ = done_tx.send(());
|
||||
})
|
||||
.expect("spawn worker");
|
||||
|
||||
match done_rx.recv_timeout(timeout) {
|
||||
Ok(()) => {}
|
||||
Err(_) => panic!("{msg}"),
|
||||
}
|
||||
worker.join().expect("worker panicked");
|
||||
}
|
||||
|
||||
fn make_watched_picker(base: &std::path::Path) -> (SharedFilePicker, SharedFrecency) {
|
||||
let shared_picker = SharedFilePicker::default();
|
||||
let shared_frecency = SharedFrecency::default();
|
||||
|
||||
FilePicker::new_with_shared_state(
|
||||
shared_picker.clone(),
|
||||
shared_frecency.clone(),
|
||||
FilePickerOptions {
|
||||
base_path: base.to_string_lossy().to_string(),
|
||||
enable_mmap_cache: false,
|
||||
enable_content_indexing: false,
|
||||
mode: FFFMode::Neovim,
|
||||
watch: true,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.expect("Failed to create FilePicker");
|
||||
|
||||
assert!(
|
||||
shared_picker.wait_for_scan(Duration::from_secs(10)),
|
||||
"initial scan never completed"
|
||||
);
|
||||
assert!(
|
||||
shared_picker.wait_for_watcher(Duration::from_secs(10)),
|
||||
"watcher never installed"
|
||||
);
|
||||
|
||||
(shared_picker, shared_frecency)
|
||||
}
|
||||
|
||||
/// Hazard (1): debouncer event handler is waiting on `shared_picker.write()`
|
||||
/// while the caller joins it from under the same guard.
|
||||
#[test]
|
||||
fn stop_background_monitor_under_write_lock_does_not_deadlock_file_events() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let base = tmp.path().to_path_buf();
|
||||
|
||||
for i in 0..4 {
|
||||
fs::write(base.join(format!("file_{i}.txt")), format!("seed {i}\n")).unwrap();
|
||||
}
|
||||
|
||||
let (shared_picker, _shared_frecency) = make_watched_picker(&base);
|
||||
|
||||
// Produce enough filesystem churn that the debouncer has events
|
||||
// queued and is likely mid-handler by the time we call stop.
|
||||
for round in 0..8 {
|
||||
for i in 0..4 {
|
||||
let path = base.join(format!("file_{i}.txt"));
|
||||
fs::write(&path, format!("edit {round}-{i}\n")).unwrap();
|
||||
}
|
||||
}
|
||||
|
||||
// Give the kernel time to deliver events into the debouncer queue
|
||||
// (50 ms = default debouncer tick).
|
||||
std::thread::sleep(Duration::from_millis(60));
|
||||
|
||||
let sp = shared_picker.clone();
|
||||
run_with_deadlock_guard(
|
||||
"stop_background_monitor() deadlocked under shared_picker.write() — \
|
||||
the debouncer thread is likely waiting on the same write lock \
|
||||
while we join it",
|
||||
Duration::from_secs(5),
|
||||
move || {
|
||||
let mut guard = sp.write().expect("write lock");
|
||||
if let Some(ref mut picker) = *guard {
|
||||
picker.stop_background_monitor();
|
||||
}
|
||||
},
|
||||
);
|
||||
}
|
||||
|
||||
/// Hazard (2): owner thread holds the debouncer mutex while waiting
|
||||
/// on `shared_picker.write()`, and `stop()` takes the debouncer mutex
|
||||
/// under the caller's write guard.
|
||||
#[test]
|
||||
fn stop_background_monitor_under_write_lock_does_not_deadlock_new_dirs() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let base = tmp.path().to_path_buf();
|
||||
|
||||
fs::write(base.join("seed.txt"), "seed\n").unwrap();
|
||||
|
||||
let (shared_picker, _shared_frecency) = make_watched_picker(&base);
|
||||
|
||||
// Create a burst of new subdirectories with files inside. On Linux
|
||||
// the watcher event thread sends each new dir to `watch_tx`, and
|
||||
// the owner thread processes them (taking the debouncer mutex +
|
||||
// `shared_picker.write()`). On macOS the owner thread still runs
|
||||
// `track_files_from_new_directories`, which takes the write lock.
|
||||
for d in 0..8 {
|
||||
let sub = base.join(format!("sub_{d}"));
|
||||
fs::create_dir(&sub).unwrap();
|
||||
for f in 0..4 {
|
||||
fs::write(sub.join(format!("f_{f}.txt")), format!("{d}-{f}\n")).unwrap();
|
||||
}
|
||||
}
|
||||
|
||||
std::thread::sleep(Duration::from_millis(120));
|
||||
|
||||
let sp = shared_picker.clone();
|
||||
run_with_deadlock_guard(
|
||||
"stop_background_monitor() deadlocked under shared_picker.write() — \
|
||||
the watcher owner thread is likely holding the debouncer mutex and \
|
||||
waiting on the same write lock while we try to take the debouncer \
|
||||
mutex to tear it down",
|
||||
Duration::from_secs(5),
|
||||
move || {
|
||||
let mut guard = sp.write().expect("write lock");
|
||||
if let Some(ref mut picker) = *guard {
|
||||
picker.stop_background_monitor();
|
||||
}
|
||||
},
|
||||
);
|
||||
}
|
||||
@@ -0,0 +1,197 @@
|
||||
#![cfg(target_os = "linux")]
|
||||
|
||||
use std::fs;
|
||||
use std::path::PathBuf;
|
||||
use std::time::{Duration, Instant};
|
||||
|
||||
use tempfile::TempDir;
|
||||
|
||||
use fff_search::file_picker::{FFFMode, FilePicker};
|
||||
use fff_search::{FilePickerOptions, SharedFilePicker, SharedFrecency};
|
||||
|
||||
/// Thread comm names Linux exposes via `/proc/self/task/*/comm` are
|
||||
/// capped at `TASK_COMM_LEN - 1 = 15` bytes. Our owner thread is named
|
||||
/// `"fff-watcher-owner"` (17 bytes), so what actually appears in
|
||||
/// `/proc` is the 15-byte truncation below.
|
||||
const WATCHER_OWNER_THREAD_NAME: &str = "fff-watcher-own";
|
||||
|
||||
/// Walk `/proc/self/task/*/comm` and return how many live threads
|
||||
/// carry `name` as their `comm`.
|
||||
fn count_live_threads_named(name: &str) -> usize {
|
||||
let Ok(dir) = fs::read_dir("/proc/self/task") else {
|
||||
return 0;
|
||||
};
|
||||
let mut count = 0usize;
|
||||
for entry in dir.flatten() {
|
||||
let comm_path = entry.path().join("comm");
|
||||
if let Ok(content) = fs::read_to_string(&comm_path) {
|
||||
if content.trim_end() == name {
|
||||
count += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
count
|
||||
}
|
||||
|
||||
/// Poll until the thread count matches `expected` or we hit `timeout`.
|
||||
fn wait_for_thread_count(name: &str, expected: usize, timeout: Duration) -> usize {
|
||||
let deadline = Instant::now() + timeout;
|
||||
loop {
|
||||
let count = count_live_threads_named(name);
|
||||
if count == expected {
|
||||
return count;
|
||||
}
|
||||
if Instant::now() >= deadline {
|
||||
return count;
|
||||
}
|
||||
std::thread::sleep(Duration::from_millis(25));
|
||||
}
|
||||
}
|
||||
|
||||
fn seed_repo(base: &std::path::Path) {
|
||||
fs::create_dir_all(base.join("src")).unwrap();
|
||||
fs::write(base.join("README.md"), "# seed\n").unwrap();
|
||||
fs::write(base.join("src/main.rs"), "fn main() {}\n").unwrap();
|
||||
fs::write(base.join("src/lib.rs"), "// lib\n").unwrap();
|
||||
|
||||
let _ = std::process::Command::new("git")
|
||||
.args(["init", "-q", "-b", "main"])
|
||||
.current_dir(base)
|
||||
.output();
|
||||
}
|
||||
|
||||
fn spawn_watched_picker(base: PathBuf) -> (SharedFilePicker, SharedFrecency) {
|
||||
let shared_picker = SharedFilePicker::default();
|
||||
let shared_frecency = SharedFrecency::default();
|
||||
|
||||
FilePicker::new_with_shared_state(
|
||||
shared_picker.clone(),
|
||||
shared_frecency.clone(),
|
||||
FilePickerOptions {
|
||||
base_path: base.to_string_lossy().to_string(),
|
||||
enable_mmap_cache: false,
|
||||
enable_content_indexing: false,
|
||||
mode: FFFMode::Neovim,
|
||||
watch: true,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.expect("FilePicker::new_with_shared_state");
|
||||
|
||||
assert!(
|
||||
shared_picker.wait_for_scan(Duration::from_secs(10)),
|
||||
"initial scan did not complete"
|
||||
);
|
||||
assert!(
|
||||
shared_picker.wait_for_watcher(Duration::from_secs(10)),
|
||||
"watcher did not install"
|
||||
);
|
||||
|
||||
(shared_picker, shared_frecency)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn watcher_threads_do_not_leak_across_picker_lifetimes() {
|
||||
// this is needed because I run this within neovim with it's own fff owner thread lmao
|
||||
let baseline = count_live_threads_named(WATCHER_OWNER_THREAD_NAME);
|
||||
|
||||
const PICKER_COUNT: usize = 4;
|
||||
|
||||
let mut tmpdirs: Vec<TempDir> = (0..PICKER_COUNT)
|
||||
.map(|_| TempDir::new().expect("mktemp"))
|
||||
.collect();
|
||||
for td in &tmpdirs {
|
||||
seed_repo(td.path());
|
||||
}
|
||||
|
||||
let mut pickers: Vec<(SharedFilePicker, SharedFrecency)> = tmpdirs
|
||||
.iter()
|
||||
.map(|td| spawn_watched_picker(td.path().canonicalize().expect("canonicalize tmp")))
|
||||
.collect();
|
||||
|
||||
let peak = wait_for_thread_count(
|
||||
WATCHER_OWNER_THREAD_NAME,
|
||||
baseline + PICKER_COUNT,
|
||||
Duration::from_secs(5),
|
||||
);
|
||||
assert_eq!(
|
||||
peak,
|
||||
baseline + PICKER_COUNT,
|
||||
"expected {} watcher-owner threads alive (baseline {} + {} pickers), saw {}",
|
||||
baseline + PICKER_COUNT,
|
||||
baseline,
|
||||
PICKER_COUNT,
|
||||
peak,
|
||||
);
|
||||
|
||||
for i in 0..PICKER_COUNT {
|
||||
let expected_remaining = baseline + PICKER_COUNT - (i + 1);
|
||||
let (sp, sf) = pickers.remove(0);
|
||||
drop(sp);
|
||||
drop(sf);
|
||||
let count = wait_for_thread_count(
|
||||
WATCHER_OWNER_THREAD_NAME,
|
||||
expected_remaining,
|
||||
Duration::from_secs(5),
|
||||
);
|
||||
assert_eq!(
|
||||
count,
|
||||
expected_remaining,
|
||||
"after dropping picker {}/{}: expected {} owner threads, saw {}",
|
||||
i + 1,
|
||||
PICKER_COUNT,
|
||||
expected_remaining,
|
||||
count,
|
||||
);
|
||||
}
|
||||
tmpdirs.clear();
|
||||
|
||||
let after_stage1 = count_live_threads_named(WATCHER_OWNER_THREAD_NAME);
|
||||
assert_eq!(
|
||||
after_stage1, baseline,
|
||||
"stage 1 leaked watcher-owner threads: baseline {}, observed {}",
|
||||
baseline, after_stage1,
|
||||
);
|
||||
|
||||
const ROUNDS: usize = 3;
|
||||
|
||||
for round in 0..ROUNDS {
|
||||
let tmp = TempDir::new().expect("mktemp");
|
||||
seed_repo(tmp.path());
|
||||
let base = tmp.path().canonicalize().expect("canonicalize tmp");
|
||||
|
||||
let (sp, sf) = spawn_watched_picker(base);
|
||||
|
||||
let during = wait_for_thread_count(
|
||||
WATCHER_OWNER_THREAD_NAME,
|
||||
baseline + 1,
|
||||
Duration::from_secs(5),
|
||||
);
|
||||
assert_eq!(
|
||||
during,
|
||||
baseline + 1,
|
||||
"round {round}: expected 1 owner thread during run, saw {during} \
|
||||
(baseline {baseline})",
|
||||
);
|
||||
|
||||
drop(sp);
|
||||
drop(sf);
|
||||
drop(tmp);
|
||||
|
||||
let after =
|
||||
wait_for_thread_count(WATCHER_OWNER_THREAD_NAME, baseline, Duration::from_secs(5));
|
||||
assert_eq!(
|
||||
after, baseline,
|
||||
"round {round}: owner thread leaked after teardown \
|
||||
(baseline {baseline}, observed {after})",
|
||||
);
|
||||
}
|
||||
|
||||
let final_count = count_live_threads_named(WATCHER_OWNER_THREAD_NAME);
|
||||
assert_eq!(
|
||||
final_count, baseline,
|
||||
"watcher-owner threads leaked past the end of the test \
|
||||
(baseline {}, final {})",
|
||||
baseline, final_count,
|
||||
);
|
||||
}
|
||||
@@ -0,0 +1,11 @@
|
||||
[package]
|
||||
name = "fff-grep"
|
||||
description = "File grepping logic for fff"
|
||||
license = "MIT"
|
||||
authors = ["Dmitriy Kovalenko <dmtr.kovalenko@outlok.com>"]
|
||||
version = "0.8.1"
|
||||
edition = "2024"
|
||||
|
||||
[dependencies]
|
||||
bstr = { version = "1.6.2", default-features = false, features = ["std"] }
|
||||
memchr = "2.6.3"
|
||||
@@ -8,10 +8,12 @@ Only `search_slice` is supported -- no file/reader/mmap search.
|
||||
#![deny(missing_docs)]
|
||||
|
||||
pub use crate::{
|
||||
matcher::{LineTerminator, Match, Matcher, NoError},
|
||||
searcher::{Searcher, SearcherBuilder},
|
||||
sink::{Sink, SinkError, SinkFinish, SinkMatch},
|
||||
};
|
||||
|
||||
pub mod lines;
|
||||
pub mod matcher;
|
||||
mod searcher;
|
||||
mod sink;
|
||||
@@ -2,10 +2,9 @@
|
||||
A collection of routines for performing operations on lines.
|
||||
*/
|
||||
|
||||
use {
|
||||
bstr::ByteSlice,
|
||||
grep_matcher::{LineTerminator, Match},
|
||||
};
|
||||
use bstr::ByteSlice;
|
||||
|
||||
use crate::matcher::{LineTerminator, Match};
|
||||
|
||||
/// An explicit iterator over lines in a particular slice of bytes.
|
||||
///
|
||||
@@ -104,7 +103,7 @@ pub fn locate(bytes: &[u8], line_term: u8, range: Match) -> Match {
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
const SHERLOCK: &'static str = "\
|
||||
const SHERLOCK: &str = "\
|
||||
For the Doctor Watsons of this world, as opposed to the Sherlock
|
||||
Holmeses, success in the province of detective work must always
|
||||
be, to a very large extent, the result of luck. Sherlock Holmes
|
||||
@@ -0,0 +1,175 @@
|
||||
//! Matcher trait inspired by ripgrep's `Matcher` just simpler
|
||||
|
||||
/// A byte range representing a match.
|
||||
#[derive(Clone, Copy, Debug, Eq, Hash, PartialEq)]
|
||||
pub struct Match {
|
||||
start: usize,
|
||||
end: usize,
|
||||
}
|
||||
|
||||
impl Match {
|
||||
/// Create a new match from start/end byte offsets.
|
||||
#[inline]
|
||||
pub fn new(start: usize, end: usize) -> Match {
|
||||
debug_assert!(start <= end);
|
||||
Match { start, end }
|
||||
}
|
||||
|
||||
/// Create a zero-width match at `offset`.
|
||||
#[inline]
|
||||
pub fn zero(offset: usize) -> Match {
|
||||
Match {
|
||||
start: offset,
|
||||
end: offset,
|
||||
}
|
||||
}
|
||||
|
||||
/// Start byte offset.
|
||||
#[inline]
|
||||
pub fn start(&self) -> usize {
|
||||
self.start
|
||||
}
|
||||
|
||||
/// End byte offset (exclusive).
|
||||
#[inline]
|
||||
pub fn end(&self) -> usize {
|
||||
self.end
|
||||
}
|
||||
|
||||
/// Return a copy with a different end offset.
|
||||
#[inline]
|
||||
pub fn with_end(&self, end: usize) -> Match {
|
||||
debug_assert!(self.start <= end);
|
||||
Match { end, ..*self }
|
||||
}
|
||||
|
||||
/// Shift both offsets forward by `amount`.
|
||||
#[inline]
|
||||
pub fn offset(&self, amount: usize) -> Match {
|
||||
Match {
|
||||
start: self.start + amount,
|
||||
end: self.end + amount,
|
||||
}
|
||||
}
|
||||
|
||||
/// Byte length of the match.
|
||||
#[inline]
|
||||
pub fn len(&self) -> usize {
|
||||
self.end - self.start
|
||||
}
|
||||
|
||||
/// True if this is a zero-width match.
|
||||
#[inline]
|
||||
pub fn is_empty(&self) -> bool {
|
||||
self.len() == 0
|
||||
}
|
||||
}
|
||||
|
||||
impl std::ops::Index<Match> for [u8] {
|
||||
type Output = [u8];
|
||||
|
||||
#[inline]
|
||||
fn index(&self, index: Match) -> &[u8] {
|
||||
&self[index.start..index.end]
|
||||
}
|
||||
}
|
||||
|
||||
impl std::ops::IndexMut<Match> for [u8] {
|
||||
#[inline]
|
||||
fn index_mut(&mut self, index: Match) -> &mut [u8] {
|
||||
&mut self[index.start..index.end]
|
||||
}
|
||||
}
|
||||
|
||||
impl std::ops::Index<Match> for str {
|
||||
type Output = str;
|
||||
|
||||
#[inline]
|
||||
fn index(&self, index: Match) -> &str {
|
||||
&self[index.start..index.end]
|
||||
}
|
||||
}
|
||||
|
||||
/// A line terminator (always a single byte for fff — no CRLF support needed).
|
||||
#[derive(Clone, Copy, Debug, Eq, Hash, PartialEq)]
|
||||
pub struct LineTerminator(u8);
|
||||
|
||||
impl LineTerminator {
|
||||
/// Create a line terminator from a single byte.
|
||||
#[inline]
|
||||
pub fn byte(byte: u8) -> LineTerminator {
|
||||
LineTerminator(byte)
|
||||
}
|
||||
|
||||
/// Return the terminator byte.
|
||||
#[inline]
|
||||
pub fn as_byte(&self) -> u8 {
|
||||
self.0
|
||||
}
|
||||
|
||||
/// Return the terminator as a single-element byte slice.
|
||||
#[inline]
|
||||
pub fn as_bytes(&self) -> &[u8] {
|
||||
std::slice::from_ref(&self.0)
|
||||
}
|
||||
}
|
||||
|
||||
impl Default for LineTerminator {
|
||||
#[inline]
|
||||
fn default() -> LineTerminator {
|
||||
LineTerminator(b'\n')
|
||||
}
|
||||
}
|
||||
|
||||
/// An error type for matchers that never produce errors.
|
||||
#[derive(Debug, Eq, PartialEq)]
|
||||
pub struct NoError(());
|
||||
|
||||
impl std::error::Error for NoError {}
|
||||
|
||||
impl std::fmt::Display for NoError {
|
||||
fn fmt(&self, _: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
unreachable!("NoError should never be instantiated")
|
||||
}
|
||||
}
|
||||
|
||||
/// A matcher finds byte-level matches in a haystack.
|
||||
pub trait Matcher {
|
||||
/// The error type (use [`NoError`] for infallible matchers).
|
||||
type Error: std::fmt::Display;
|
||||
|
||||
/// Find the first match at or after `at` in `haystack`.
|
||||
fn find_at(&self, haystack: &[u8], at: usize) -> Result<Option<Match>, Self::Error>;
|
||||
|
||||
/// Find the first match in `haystack`.
|
||||
#[inline]
|
||||
fn find(&self, haystack: &[u8]) -> Result<Option<Match>, Self::Error> {
|
||||
self.find_at(haystack, 0)
|
||||
}
|
||||
|
||||
/// The line terminator this matcher guarantees will never appear in a match.
|
||||
/// Return `None` if the matcher can match across lines.
|
||||
#[inline]
|
||||
fn line_terminator(&self) -> Option<LineTerminator> {
|
||||
None
|
||||
}
|
||||
}
|
||||
|
||||
impl<M: Matcher> Matcher for &M {
|
||||
type Error = M::Error;
|
||||
|
||||
#[inline]
|
||||
fn find_at(&self, haystack: &[u8], at: usize) -> Result<Option<Match>, Self::Error> {
|
||||
(*self).find_at(haystack, at)
|
||||
}
|
||||
|
||||
#[inline]
|
||||
fn find(&self, haystack: &[u8]) -> Result<Option<Match>, Self::Error> {
|
||||
(*self).find(haystack)
|
||||
}
|
||||
|
||||
#[inline]
|
||||
fn line_terminator(&self) -> Option<LineTerminator> {
|
||||
(*self).line_terminator()
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,139 @@
|
||||
use crate::{
|
||||
lines,
|
||||
matcher::Matcher,
|
||||
searcher::{Config, Range, Searcher},
|
||||
sink::{Sink, SinkError, SinkFinish, SinkMatch},
|
||||
};
|
||||
|
||||
#[derive(Debug)]
|
||||
pub(crate) struct Core<'s, M: 's, S> {
|
||||
config: &'s Config,
|
||||
matcher: M,
|
||||
searcher: &'s Searcher,
|
||||
sink: S,
|
||||
pos: usize,
|
||||
absolute_byte_offset: u64,
|
||||
line_number: Option<u64>,
|
||||
last_line_counted: usize,
|
||||
last_line_visited: usize,
|
||||
}
|
||||
|
||||
impl<'s, M: Matcher, S: Sink> Core<'s, M, S> {
|
||||
pub(crate) fn new(searcher: &'s Searcher, matcher: M, sink: S) -> Core<'s, M, S> {
|
||||
let line_number = if searcher.config.line_number {
|
||||
Some(1)
|
||||
} else {
|
||||
None
|
||||
};
|
||||
Core {
|
||||
config: &searcher.config,
|
||||
matcher,
|
||||
searcher,
|
||||
sink,
|
||||
pos: 0,
|
||||
absolute_byte_offset: 0,
|
||||
line_number,
|
||||
last_line_counted: 0,
|
||||
last_line_visited: 0,
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn pos(&self) -> usize {
|
||||
self.pos
|
||||
}
|
||||
|
||||
pub(crate) fn set_pos(&mut self, pos: usize) {
|
||||
self.pos = pos;
|
||||
}
|
||||
|
||||
pub(crate) fn matched(&mut self, buf: &[u8], range: &Range) -> Result<bool, S::Error> {
|
||||
self.sink_matched(buf, range)
|
||||
}
|
||||
|
||||
pub(crate) fn find(&mut self, slice: &[u8]) -> Result<Option<Range>, S::Error> {
|
||||
self.matcher.find(slice).map_err(S::Error::error_message)
|
||||
}
|
||||
|
||||
pub(crate) fn begin(&mut self) -> Result<bool, S::Error> {
|
||||
self.sink.begin(self.searcher)
|
||||
}
|
||||
|
||||
pub(crate) fn finish(&mut self, byte_count: u64) -> Result<(), S::Error> {
|
||||
self.sink.finish(self.searcher, &SinkFinish { byte_count })
|
||||
}
|
||||
|
||||
pub(crate) fn match_by_line(&mut self, buf: &[u8]) -> Result<bool, S::Error> {
|
||||
while !buf[self.pos()..].is_empty() {
|
||||
if let Some(line) = self.find_by_line(buf)? {
|
||||
self.set_pos(line.end());
|
||||
if !self.sink_matched(buf, &line)? {
|
||||
return Ok(false);
|
||||
}
|
||||
} else {
|
||||
break;
|
||||
}
|
||||
}
|
||||
self.set_pos(buf.len());
|
||||
Ok(true)
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
fn find_by_line(&mut self, buf: &[u8]) -> Result<Option<Range>, S::Error> {
|
||||
let mut pos = self.pos();
|
||||
while !buf[pos..].is_empty() {
|
||||
let mat = match self
|
||||
.matcher
|
||||
.find(&buf[pos..])
|
||||
.map_err(S::Error::error_message)?
|
||||
{
|
||||
None => return Ok(None),
|
||||
Some(m) => m,
|
||||
};
|
||||
let line = lines::locate(
|
||||
buf,
|
||||
self.config.line_term.as_byte(),
|
||||
Range::zero(mat.start()).offset(pos),
|
||||
);
|
||||
if line.start() == buf.len() {
|
||||
pos = buf.len();
|
||||
continue;
|
||||
}
|
||||
return Ok(Some(line));
|
||||
}
|
||||
Ok(None)
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
fn sink_matched(&mut self, buf: &[u8], range: &Range) -> Result<bool, S::Error> {
|
||||
self.count_lines(buf, range.start());
|
||||
let offset = self.absolute_byte_offset + range.start() as u64;
|
||||
let linebuf = &buf[*range];
|
||||
let keepgoing = self.sink.matched(
|
||||
self.searcher,
|
||||
&SinkMatch {
|
||||
bytes: linebuf,
|
||||
absolute_byte_offset: offset,
|
||||
line_number: self.line_number,
|
||||
buffer: buf,
|
||||
bytes_range_in_buffer: range.start()..range.end(),
|
||||
},
|
||||
)?;
|
||||
if !keepgoing {
|
||||
return Ok(false);
|
||||
}
|
||||
self.last_line_visited = range.end();
|
||||
Ok(true)
|
||||
}
|
||||
|
||||
fn count_lines(&mut self, buf: &[u8], upto: usize) {
|
||||
if let Some(ref mut line_number) = self.line_number {
|
||||
if self.last_line_counted >= upto {
|
||||
return;
|
||||
}
|
||||
let slice = &buf[self.last_line_counted..upto];
|
||||
let count = lines::count(slice, self.config.line_term.as_byte());
|
||||
*line_number += count;
|
||||
self.last_line_counted = upto;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,7 +1,6 @@
|
||||
use grep_matcher::Matcher;
|
||||
|
||||
use crate::{
|
||||
lines,
|
||||
matcher::Matcher,
|
||||
searcher::{Config, Range, Searcher, core::Core},
|
||||
sink::Sink,
|
||||
};
|
||||
@@ -1,6 +1,5 @@
|
||||
use grep_matcher::{LineTerminator, Match, Matcher};
|
||||
|
||||
use crate::{
|
||||
matcher::{LineTerminator, Match, Matcher},
|
||||
searcher::glue::{MultiLine, SliceByLine},
|
||||
sink::{Sink, SinkError},
|
||||
};
|
||||
@@ -120,12 +119,7 @@ impl Searcher {
|
||||
|
||||
/// Execute a search over the given slice and write the results to the
|
||||
/// given sink.
|
||||
pub fn search_slice<M, S>(
|
||||
&mut self,
|
||||
matcher: M,
|
||||
slice: &[u8],
|
||||
write_to: S,
|
||||
) -> Result<(), S::Error>
|
||||
pub fn search_slice<M, S>(&self, matcher: M, slice: &[u8], write_to: S) -> Result<(), S::Error>
|
||||
where
|
||||
M: Matcher,
|
||||
S: Sink,
|
||||
@@ -195,11 +189,6 @@ impl Searcher {
|
||||
{
|
||||
return false;
|
||||
}
|
||||
if let Some(non_matching) = matcher.non_matching_bytes()
|
||||
&& non_matching.contains(self.line_terminator().as_byte())
|
||||
{
|
||||
return false;
|
||||
}
|
||||
true
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,27 @@
|
||||
[package]
|
||||
name = "fff-mcp"
|
||||
version = "0.8.1"
|
||||
edition = "2024"
|
||||
description = "MCP server for FFF file finder - drop-in replacement for AI code assistant search tools"
|
||||
license = "MIT"
|
||||
|
||||
[[bin]]
|
||||
name = "fff-mcp"
|
||||
path = "src/main.rs"
|
||||
|
||||
[features]
|
||||
default = ["zlob"]
|
||||
zlob = ["fff/zlob"]
|
||||
|
||||
[dependencies]
|
||||
fff = { package = "fff-search", path = "../fff-core", default-features = false , version = "0.8.1" }
|
||||
fff-query-parser = { path = "../fff-query-parser", default-features = false , version = "0.8.1" }
|
||||
mimalloc = { workspace = true }
|
||||
rmcp = { version = "1.1.0", features = ["server", "transport-io"] }
|
||||
schemars = "1.0"
|
||||
serde = { version = "1.0", features = ["derive"] }
|
||||
serde_json = "1.0"
|
||||
tokio = { version = "1", features = ["full"] }
|
||||
tracing = { workspace = true }
|
||||
git2 = { workspace = true }
|
||||
clap = { version = "4", features = ["derive", "env"] }
|
||||
@@ -0,0 +1,15 @@
|
||||
fn main() {
|
||||
// Embed the git commit hash at build time for update checking.
|
||||
let hash = std::process::Command::new("git")
|
||||
.args(["rev-parse", "HEAD"])
|
||||
.output()
|
||||
.ok()
|
||||
.filter(|o| o.status.success())
|
||||
.and_then(|o| String::from_utf8(o.stdout).ok())
|
||||
.map(|s| s.trim().to_string())
|
||||
.unwrap_or_else(|| "unknown".to_string());
|
||||
|
||||
println!("cargo:rustc-env=FFF_GIT_HASH={}", hash);
|
||||
println!("cargo:rerun-if-changed=../../.git/HEAD");
|
||||
println!("cargo:rerun-if-changed=../../.git/refs/");
|
||||
}
|
||||
@@ -0,0 +1,52 @@
|
||||
//! Cursor store for grep pagination.
|
||||
//!
|
||||
//! Maintains an in-memory map of opaque cursor IDs to file offsets.
|
||||
//! Cursors are evicted LRU-style when the store exceeds capacity.
|
||||
|
||||
use std::collections::{HashMap, VecDeque};
|
||||
|
||||
const MAX_CURSORS: usize = 20;
|
||||
|
||||
/// Stores cursor state for paginated grep results.
|
||||
pub struct CursorStore {
|
||||
counter: u64,
|
||||
/// Map from cursor ID string → file offset for next page.
|
||||
cursors: HashMap<String, usize>,
|
||||
/// Insertion order for LRU eviction.
|
||||
insertion_order: VecDeque<String>,
|
||||
}
|
||||
|
||||
impl CursorStore {
|
||||
pub fn new() -> Self {
|
||||
Self {
|
||||
counter: 0,
|
||||
cursors: HashMap::new(),
|
||||
insertion_order: VecDeque::new(),
|
||||
}
|
||||
}
|
||||
|
||||
/// Store a cursor and return its opaque ID string.
|
||||
pub fn store(&mut self, file_offset: usize) -> String {
|
||||
self.counter = self.counter.wrapping_add(1);
|
||||
let id = self.counter.to_string();
|
||||
|
||||
self.cursors.insert(id.clone(), file_offset);
|
||||
self.insertion_order.push_back(id.clone());
|
||||
|
||||
// Evict oldest cursors
|
||||
while self.cursors.len() > MAX_CURSORS {
|
||||
if let Some(oldest) = self.insertion_order.pop_front() {
|
||||
self.cursors.remove(&oldest);
|
||||
} else {
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
id
|
||||
}
|
||||
|
||||
/// Retrieve the file offset for a cursor ID.
|
||||
pub fn get(&self, id: &str) -> Option<usize> {
|
||||
self.cursors.get(id).copied()
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,117 @@
|
||||
use crate::Args;
|
||||
use git2::Repository;
|
||||
|
||||
fn check(label: &str, ok: bool, detail: &str) -> bool {
|
||||
let marker = if ok { "+" } else { "x" };
|
||||
println!(" [{marker}] {label}: {detail}");
|
||||
ok
|
||||
}
|
||||
|
||||
fn warn(label: &str, detail: &str) {
|
||||
println!(" [!] {label}: {detail}");
|
||||
}
|
||||
|
||||
pub fn run_healthcheck(args: &Args) -> Result<(), Box<dyn std::error::Error>> {
|
||||
let version = concat!(env!("CARGO_PKG_VERSION"), " (", env!("FFF_GIT_HASH"), ")");
|
||||
println!("fff-mcp {version}\n");
|
||||
|
||||
let mut all_ok = true;
|
||||
|
||||
// 1. Base path
|
||||
let base_path = args.base_path.clone().unwrap_or_else(|| {
|
||||
std::env::current_dir()
|
||||
.unwrap_or_default()
|
||||
.to_string_lossy()
|
||||
.to_string()
|
||||
});
|
||||
|
||||
let path_exists = std::path::Path::new(&base_path).is_dir();
|
||||
all_ok &= check(
|
||||
"Base path",
|
||||
path_exists,
|
||||
if path_exists {
|
||||
&base_path
|
||||
} else {
|
||||
"directory does not exist"
|
||||
},
|
||||
);
|
||||
|
||||
// 2. Git repository
|
||||
match Repository::discover(&base_path) {
|
||||
Ok(repo) => {
|
||||
if let Some(workdir) = repo.workdir() {
|
||||
all_ok &= check("Git repository", true, &format!("{}", workdir.display()));
|
||||
} else {
|
||||
all_ok &= check("Git repository", true, "bare repository");
|
||||
}
|
||||
}
|
||||
Err(_) => {
|
||||
// Not fatal — fff-mcp works without git, but worth flagging.
|
||||
warn(
|
||||
"Git repository",
|
||||
"not found (fff-mcp will still work, but git-status features are disabled)",
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
// 3. Frecency database
|
||||
if let Some(ref db_path) = args.frecency_db_path {
|
||||
let parent_ok = std::path::Path::new(db_path)
|
||||
.parent()
|
||||
.is_some_and(|p| p.is_dir());
|
||||
all_ok &= check(
|
||||
"Frecency DB",
|
||||
parent_ok,
|
||||
if parent_ok {
|
||||
db_path
|
||||
} else {
|
||||
"parent directory does not exist"
|
||||
},
|
||||
);
|
||||
} else {
|
||||
check("Frecency DB", false, "path not resolved");
|
||||
}
|
||||
|
||||
// 4. Query history database
|
||||
if let Some(ref db_path) = args.history_db_path {
|
||||
let parent_ok = std::path::Path::new(db_path)
|
||||
.parent()
|
||||
.is_some_and(|p| p.is_dir());
|
||||
all_ok &= check(
|
||||
"History DB",
|
||||
parent_ok,
|
||||
if parent_ok {
|
||||
db_path
|
||||
} else {
|
||||
"parent directory does not exist"
|
||||
},
|
||||
);
|
||||
} else {
|
||||
check("History DB", false, "path not resolved");
|
||||
}
|
||||
|
||||
// 5. Log file
|
||||
if let Some(ref log_path) = args.log_file {
|
||||
let parent_ok = std::path::Path::new(log_path)
|
||||
.parent()
|
||||
.is_some_and(|p| p.is_dir());
|
||||
all_ok &= check(
|
||||
"Log file",
|
||||
parent_ok,
|
||||
if parent_ok {
|
||||
log_path
|
||||
} else {
|
||||
"parent directory does not exist"
|
||||
},
|
||||
);
|
||||
} else {
|
||||
check("Log file", false, "path not resolved");
|
||||
}
|
||||
|
||||
if all_ok {
|
||||
println!("All checks passed.");
|
||||
Ok(())
|
||||
} else {
|
||||
Err("Some checks failed — review the items marked [x] above.".into())
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,328 @@
|
||||
//! FFF MCP Server — high-performance file finder for AI code assistants.
|
||||
//!
|
||||
//! Drop-in replacement for AI code assistant file search tools (Glob/Grep).
|
||||
//! Provides frecency-ranked, fuzzy-matched, git-aware file finding and
|
||||
//! code search via the Model Context Protocol (MCP).
|
||||
//!
|
||||
//! Uses `fff-core` directly (zero FFI overhead) for all search operations.
|
||||
|
||||
mod cursor;
|
||||
mod healthcheck;
|
||||
mod output;
|
||||
mod server;
|
||||
mod update_check;
|
||||
|
||||
use clap::Parser;
|
||||
use fff::file_picker::FilePicker;
|
||||
use fff::frecency::FrecencyTracker;
|
||||
use fff::{FFFMode, SharedFilePicker, SharedFrecency};
|
||||
use git2::Repository;
|
||||
use mimalloc::MiMalloc;
|
||||
use rmcp::{ServiceExt, transport::stdio};
|
||||
use server::FffServer;
|
||||
|
||||
#[global_allocator]
|
||||
static GLOBAL: MiMalloc = MiMalloc;
|
||||
|
||||
pub const MCP_INSTRUCTIONS: &str = concat!(
|
||||
"FFF is a fast file finder with frecency-ranked results (frequent/recent files first, git-dirty files boosted).\n",
|
||||
"\n",
|
||||
"## Which Tool Should I Use?\n",
|
||||
"\n",
|
||||
"- **grep**: DEFAULT tool. Searches file CONTENTS -- definitions, usage, patterns. Use when you have a specific name or pattern.\n",
|
||||
"- **find_files**: Explores which files/modules exist for a topic. Use when you DON'T have a specific identifier or LOOKING FOR A FILE.\n",
|
||||
"- **multi_grep**: OR logic across multiple patterns. Use for case variants (e.g. ['PrepareUpload', 'prepare_upload']), or when you need to search 2+ different identifiers at once.\n",
|
||||
"\n",
|
||||
"## Core Rules\n",
|
||||
"\n",
|
||||
"### 1. Search BARE IDENTIFIERS only\n",
|
||||
"Grep matches single lines. Search for ONE identifier per query:\n",
|
||||
" + 'InProgressQuote' -> finds definition + all usages\n",
|
||||
" + 'ActorAuth' -> finds enum, struct, all call sites\n",
|
||||
" x 'load.*metadata.*InProgressQuote' -> regex spanning multiple tokens, 0 results\n",
|
||||
" x 'ctx.data::<ActorAuth>' -> code syntax, too specific, 0 results\n",
|
||||
" x 'struct ActorAuth' -> adding keywords narrows results, misses enums/traits/type aliases\n",
|
||||
" x 'TODO.*#\\d+' -> complex regex, use simple 'TODO' then filter visually\n",
|
||||
"\n",
|
||||
"### 2. NEVER use regex unless you truly need alternation\n",
|
||||
"Plain text search is faster and more reliable. Regex patterns like `.*`, `\\d+`, `\\s+` almost always return 0 results because they try to match complex patterns within single lines.\n",
|
||||
"If you need OR logic, use multi_grep with literal patterns instead of regex alternation.\n",
|
||||
"\n",
|
||||
"### 3. Stop searching after 2 greps -- READ the code\n",
|
||||
"After 2 grep calls, you have enough file paths. Read the top result to understand the code.\n",
|
||||
"Do NOT keep grepping with variations. More greps != better understanding.\n",
|
||||
"\n",
|
||||
"### 4. Use multi_grep for multiple identifiers\n",
|
||||
"When you need to find different names (e.g. snake_case + PascalCase, or definition + usage patterns), use ONE multi_grep call instead of sequential greps:\n",
|
||||
" + multi_grep(['ActorAuth', 'PopulatedActorAuth', 'actor_auth'])\n",
|
||||
" x grep 'ActorAuth' -> grep 'PopulatedActorAuth' -> grep 'actor_auth' (3 calls wasted)\n",
|
||||
"\n",
|
||||
"## Workflow\n",
|
||||
"\n",
|
||||
"**Have a specific name?** -> grep the bare identifier.\n",
|
||||
"**Need multiple name variants?** -> multi_grep with all variants in one call.\n",
|
||||
"**Exploring a topic / finding files?** -> find_files.\n",
|
||||
"**Got results?** -> Read the top file. Don't grep again.\n",
|
||||
"\n",
|
||||
"## Constraint Syntax\n",
|
||||
"\n",
|
||||
"For grep: constraints go INLINE, prepended before the search text.\n",
|
||||
"For multi_grep: constraints go in the separate 'constraints' parameter.\n",
|
||||
"\n",
|
||||
"Constraints MUST match one of these formats:\n",
|
||||
" Extension: '*.rs', '*.{ts,tsx}'\n",
|
||||
" Directory: 'src/', 'quotes/'\n",
|
||||
" Filename: 'schema.rs', 'src/main.rs'\n",
|
||||
" Exclude: '!test/', '!*.spec.ts'\n",
|
||||
"\n",
|
||||
"! Bare words without extensions are NOT constraints. 'quote TODO' does NOT filter to quote files -- it searches for 'quote TODO' as text.\n",
|
||||
" + 'schema.rs TODO' -> searches for 'TODO' in files schema.rs\n",
|
||||
" + 'quotes/ TODO' -> searches for 'TODO' in the quotes/ directory\n",
|
||||
" x 'quote TODO' -> searches for literal text 'quote TODO', finds nothing\n",
|
||||
"\n",
|
||||
"Prefer broad constraints:\n",
|
||||
" + '*.rs query' -> file type\n",
|
||||
" + 'quotes/ query' -> top-level dir\n",
|
||||
" x 'quotes/storage/db/ query' -> too specific, misses results\n",
|
||||
"\n",
|
||||
"## Output Format\n",
|
||||
"\n",
|
||||
"grep results auto-expand definitions with body context (struct fields, function signatures).\n",
|
||||
"This often provides enough information WITHOUT a follow-up Read call.\n",
|
||||
"Lines marked with | are definition body context. [def] marks definition files.\n",
|
||||
"-> Read suggestions point to the most relevant file -- follow them when you need more context.\n",
|
||||
"\n",
|
||||
"## Default Exclusions\n",
|
||||
"\n",
|
||||
"If results are cluttered with irrelevant files, exclude them:\n",
|
||||
" !tests/ - exclude tests directory\n",
|
||||
" !*.spec.ts - exclude test files\n",
|
||||
" !generated/ - exclude generated code",
|
||||
);
|
||||
|
||||
/// FFF MCP Server — high-performance file finder for AI code assistants.
|
||||
#[derive(Parser)]
|
||||
#[command(name = "fff-mcp", version = concat!(env!("CARGO_PKG_VERSION"), " (", env!("FFF_GIT_HASH"), ")"))]
|
||||
pub(crate) struct Args {
|
||||
/// Base directory to index. Defaults to the current working directory.
|
||||
#[arg(value_name = "PATH")]
|
||||
base_path: Option<String>,
|
||||
|
||||
/// Path to the frecency database.
|
||||
#[arg(long = "frecency-db")]
|
||||
frecency_db_path: Option<String>,
|
||||
|
||||
/// Path to the query history database.
|
||||
#[arg(long = "history-db")]
|
||||
#[allow(dead_code)]
|
||||
history_db_path: Option<String>,
|
||||
|
||||
/// Path to the log file.
|
||||
#[arg(long = "log-file")]
|
||||
log_file: Option<String>,
|
||||
|
||||
/// Log level (e.g. trace, debug, info, warn, error).
|
||||
#[arg(long = "log-level")]
|
||||
log_level: Option<String>,
|
||||
|
||||
/// Disable automatic update checks on startup.
|
||||
#[arg(long = "no-update-check")]
|
||||
no_update_check: bool,
|
||||
|
||||
/// Disable eager mmap warmup after the initial scan. Grep results will
|
||||
/// still work (files are mmap'd lazily on first access), but the first
|
||||
/// search may be slightly slower. Useful on very large repos where the
|
||||
/// warmup would consume too many kernel resources.
|
||||
#[arg(long = "no-warmup")]
|
||||
no_warmup: bool,
|
||||
|
||||
/// Disable the content index built after the initial scan.
|
||||
/// This makes grep calls slower but consumes less RAM (recommended to not turn off)
|
||||
no_content_indexing: bool,
|
||||
|
||||
/// Explicitly enable content indexing even when `--no-warmup` is set.
|
||||
#[arg(long = "content-indexing")]
|
||||
content_indexing: bool,
|
||||
|
||||
/// Disable the background file-system watcher. Files are scanned once
|
||||
/// at startup but not monitored for changes.
|
||||
#[arg(long = "no-watch")]
|
||||
no_watch: bool,
|
||||
|
||||
/// Maximum number of files whose content is kept persistently in memory.
|
||||
/// Files beyond this limit are still searchable via temporary mmaps that
|
||||
/// are released after each grep. Defaults to 30 000.
|
||||
/// Also settable via the FFF_MAX_CACHED_FILES environment variable.
|
||||
#[arg(long = "max-cached-files", env = "FFF_MAX_CACHED_FILES")]
|
||||
max_cached_files: Option<usize>,
|
||||
|
||||
/// Run a health check and print diagnostic information, then exit.
|
||||
#[arg(long = "healthcheck")]
|
||||
pub(crate) healthcheck: bool,
|
||||
}
|
||||
|
||||
/// Resolve default paths for the log file.
|
||||
/// Database paths (frecency, history) must be explicitly provided via flags.
|
||||
fn resolve_defaults(args: &mut Args) {
|
||||
// Ensure parent directories exist for database paths when provided
|
||||
for path in [&args.frecency_db_path, &args.history_db_path]
|
||||
.into_iter()
|
||||
.flatten()
|
||||
{
|
||||
if let Some(parent) = std::path::Path::new(path).parent() {
|
||||
let _ = std::fs::create_dir_all(parent);
|
||||
}
|
||||
}
|
||||
|
||||
if args.log_file.is_none() {
|
||||
let home = dirs_home();
|
||||
let is_windows = cfg!(target_os = "windows");
|
||||
args.log_file = Some(if is_windows {
|
||||
format!("{}\\AppData\\Local\\fff_mcp.log", home)
|
||||
} else {
|
||||
format!("{}/.cache/fff_mcp.log", home)
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
fn dirs_home() -> String {
|
||||
std::env::var("HOME")
|
||||
.or_else(|_| std::env::var("USERPROFILE"))
|
||||
.unwrap_or_else(|_| "/tmp".to_string())
|
||||
}
|
||||
|
||||
#[tokio::main]
|
||||
async fn main() -> Result<(), Box<dyn std::error::Error>> {
|
||||
let mut args = Args::parse();
|
||||
resolve_defaults(&mut args);
|
||||
|
||||
if args.healthcheck {
|
||||
return healthcheck::run_healthcheck(&args);
|
||||
}
|
||||
|
||||
let log_file = args.log_file.as_deref().unwrap_or("");
|
||||
if let Err(e) = fff::log::init_tracing(log_file, args.log_level.as_deref()) {
|
||||
eprintln!("Warning: Failed to init tracing: {}", e);
|
||||
}
|
||||
|
||||
let base_path = args.base_path.unwrap_or_else(|| {
|
||||
std::env::current_dir()
|
||||
.unwrap_or_default()
|
||||
.to_string_lossy()
|
||||
.to_string()
|
||||
});
|
||||
|
||||
let base_path = match Repository::discover(&base_path) {
|
||||
Ok(repo) => {
|
||||
if let Some(workdir) = repo.workdir() {
|
||||
let git_root = workdir.to_string_lossy().to_string();
|
||||
tracing::info!("Discovered git root: {}", git_root);
|
||||
git_root
|
||||
} else {
|
||||
tracing::info!("Git repository is bare, using base path: {}", base_path);
|
||||
base_path
|
||||
}
|
||||
}
|
||||
Err(_) => {
|
||||
tracing::info!(
|
||||
"No git repository found, indexing from base path: {}",
|
||||
base_path
|
||||
);
|
||||
base_path
|
||||
}
|
||||
};
|
||||
|
||||
let shared_picker = SharedFilePicker::default();
|
||||
let shared_frecency = SharedFrecency::default();
|
||||
if let Some(frecency_db_path) = args.frecency_db_path {
|
||||
match FrecencyTracker::open(&frecency_db_path) {
|
||||
Ok(tracker) => {
|
||||
let _ = shared_frecency.init(tracker);
|
||||
}
|
||||
Err(e) => {
|
||||
eprintln!("Warning: Failed to init frecency db: {}", e);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Content indexing follows warmup by default (backward compat), unless
|
||||
// the user explicitly opts in via --content-indexing or out via
|
||||
// --no-content-indexing.
|
||||
let enable_content_indexing = if args.content_indexing {
|
||||
true
|
||||
} else if args.no_content_indexing {
|
||||
false
|
||||
} else {
|
||||
!args.no_warmup
|
||||
};
|
||||
|
||||
// Initialize file picker (spawns background scan + watcher)
|
||||
FilePicker::new_with_shared_state(
|
||||
shared_picker.clone(),
|
||||
shared_frecency.clone(),
|
||||
fff::FilePickerOptions {
|
||||
base_path,
|
||||
enable_mmap_cache: !args.no_warmup,
|
||||
enable_content_indexing,
|
||||
watch: !args.no_watch,
|
||||
mode: FFFMode::Ai,
|
||||
cache_budget: args
|
||||
.max_cached_files
|
||||
.map(fff::ContentCacheBudget::new_for_repo),
|
||||
follow_symlinks: false,
|
||||
},
|
||||
)
|
||||
.map_err(|e| format!("Failed to init file picker: {}", e))?;
|
||||
|
||||
if !args.no_update_check {
|
||||
update_check::spawn_update_check();
|
||||
}
|
||||
|
||||
// Create and start the MCP server
|
||||
let server = FffServer::new(shared_picker.clone(), shared_frecency.clone());
|
||||
|
||||
// Wait for initial scan in background — don't block server startup
|
||||
let picker_clone_for_scan = shared_picker.clone();
|
||||
tokio::task::spawn_blocking(move || {
|
||||
let start = std::time::Instant::now();
|
||||
loop {
|
||||
let is_scanning = picker_clone_for_scan
|
||||
.read()
|
||||
.ok()
|
||||
.and_then(|g| g.as_ref().map(|p| p.is_scan_active()))
|
||||
.unwrap_or(true);
|
||||
|
||||
if !is_scanning {
|
||||
tracing::info!("Initial scan completed in {:?}", start.elapsed());
|
||||
break;
|
||||
}
|
||||
std::thread::sleep(std::time::Duration::from_millis(50));
|
||||
}
|
||||
});
|
||||
|
||||
let service = server
|
||||
.serve(stdio())
|
||||
.await
|
||||
.map_err(|e| format!("Failed to start MCP server: {}", e))?;
|
||||
|
||||
let picker_for_shutdown = shared_picker.clone();
|
||||
tokio::spawn(async move {
|
||||
tokio::signal::ctrl_c().await.ok();
|
||||
if let Ok(mut guard) = picker_for_shutdown.write()
|
||||
&& let Some(ref mut picker) = *guard
|
||||
{
|
||||
picker.stop_background_monitor();
|
||||
}
|
||||
std::process::exit(0);
|
||||
});
|
||||
|
||||
service.waiting().await?;
|
||||
|
||||
if let Ok(mut guard) = shared_picker.write()
|
||||
&& let Some(ref mut picker) = *guard
|
||||
{
|
||||
picker.stop_background_monitor();
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
@@ -0,0 +1,576 @@
|
||||
//! Output formatting for MCP grep/search results.
|
||||
|
||||
use fff::GrepMatch;
|
||||
use fff::file_picker::FilePicker;
|
||||
use fff::git::format_git_status_opt;
|
||||
use fff::grep::is_import_line;
|
||||
use fff::types::FileItem;
|
||||
|
||||
use crate::cursor::CursorStore;
|
||||
|
||||
fn frecency_word(score: i32) -> Option<&'static str> {
|
||||
if score >= 100 {
|
||||
Some("hot")
|
||||
} else if score >= 50 {
|
||||
Some("warm")
|
||||
} else if score >= 10 {
|
||||
Some("frequent")
|
||||
} else {
|
||||
None
|
||||
}
|
||||
}
|
||||
|
||||
pub fn file_suffix(git_status: Option<git2::Status>, frecency_score: i32) -> String {
|
||||
match (
|
||||
frecency_word(frecency_score),
|
||||
format_git_status_opt(git_status),
|
||||
) {
|
||||
(Some(f), Some(g)) => format!(" - {f} git:{g}"),
|
||||
(Some(f), None) => format!(" - {f}"),
|
||||
(None, Some(g)) => format!(" git:{g}"),
|
||||
(None, None) => String::new(),
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum OutputMode {
|
||||
Content,
|
||||
FilesWithMatches,
|
||||
Count,
|
||||
Usage,
|
||||
}
|
||||
|
||||
impl OutputMode {
|
||||
pub fn new(s: Option<&str>) -> Self {
|
||||
match s {
|
||||
Some("files_with_matches") => Self::FilesWithMatches,
|
||||
Some("count") => Self::Count,
|
||||
Some("usage") => Self::Usage,
|
||||
_ => Self::Content,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const LARGE_FILE_BYTES: u64 = 20_000;
|
||||
|
||||
fn size_tag(bytes: u64) -> String {
|
||||
if bytes < LARGE_FILE_BYTES {
|
||||
String::new()
|
||||
} else {
|
||||
let kb = (bytes + 512) / 1024; // round
|
||||
format!(" ({}KB - use offset to read relevant section)", kb)
|
||||
}
|
||||
}
|
||||
|
||||
const MAX_PREVIEW: usize = 120;
|
||||
const MAX_LINE_LEN: usize = 180;
|
||||
const MAX_DEF_EXPAND_FIRST: usize = 8;
|
||||
const MAX_DEF_EXPAND: usize = 5;
|
||||
const MAX_FIRST_MATCH_EXPAND: usize = 8;
|
||||
|
||||
fn trauncate_line_for_ai(
|
||||
line: &str,
|
||||
match_ranges: Option<&[(u32, u32)]>,
|
||||
max_len: usize,
|
||||
) -> String {
|
||||
// Leading whitespace is already stripped by core (trim_whitespace option).
|
||||
// Only strip trailing whitespace here.
|
||||
let trimmed = line.trim_end();
|
||||
if trimmed.is_empty() {
|
||||
return String::new();
|
||||
}
|
||||
|
||||
if trimmed.len() <= max_len {
|
||||
return trimmed.to_string();
|
||||
}
|
||||
|
||||
// Use first match range to center the window
|
||||
if let Some(ranges) = match_ranges
|
||||
&& let Some(&(match_start, match_end)) = ranges.first()
|
||||
{
|
||||
let match_start = match_start as usize;
|
||||
let match_end = match_end as usize;
|
||||
let match_len = match_end.saturating_sub(match_start);
|
||||
|
||||
let budget = max_len.saturating_sub(match_len);
|
||||
let before = budget / 3;
|
||||
let after = budget - before;
|
||||
|
||||
let win_start = match_start.saturating_sub(before);
|
||||
let win_end = (match_end + after).min(trimmed.len());
|
||||
|
||||
// Clamp to char boundaries
|
||||
let win_start = floor_char_boundary(trimmed, win_start);
|
||||
let win_end = ceil_char_boundary(trimmed, win_end);
|
||||
|
||||
let mut result = trimmed[win_start..win_end].to_string();
|
||||
if win_start > 0 {
|
||||
result.insert(0, '…');
|
||||
}
|
||||
if win_end < trimmed.len() {
|
||||
result.push('…');
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
// No match ranges — truncate from start
|
||||
let end = ceil_char_boundary(trimmed, max_len);
|
||||
format!("{}…", &trimmed[..end])
|
||||
}
|
||||
|
||||
fn floor_char_boundary(s: &str, index: usize) -> usize {
|
||||
if index >= s.len() {
|
||||
return s.len();
|
||||
}
|
||||
let mut i = index;
|
||||
while i > 0 && !s.is_char_boundary(i) {
|
||||
i -= 1;
|
||||
}
|
||||
i
|
||||
}
|
||||
|
||||
fn ceil_char_boundary(s: &str, index: usize) -> usize {
|
||||
if index >= s.len() {
|
||||
return s.len();
|
||||
}
|
||||
let mut i = index;
|
||||
while i < s.len() && !s.is_char_boundary(i) {
|
||||
i += 1;
|
||||
}
|
||||
i
|
||||
}
|
||||
|
||||
struct FileMeta<'a> {
|
||||
file: &'a FileItem,
|
||||
line_number: u64,
|
||||
line_content: String,
|
||||
is_definition: bool,
|
||||
match_ranges: Vec<(u32, u32)>,
|
||||
context_after: Vec<String>,
|
||||
}
|
||||
|
||||
pub struct GrepFormatter<'a> {
|
||||
pub matches: &'a [GrepMatch],
|
||||
pub files: &'a [&'a FileItem],
|
||||
pub total_matched: usize,
|
||||
pub next_file_offset: usize,
|
||||
pub output_mode: OutputMode,
|
||||
pub max_results: usize,
|
||||
pub show_context: bool,
|
||||
pub auto_expand_defs: bool,
|
||||
pub picker: &'a FilePicker,
|
||||
}
|
||||
|
||||
impl GrepFormatter<'_> {
|
||||
pub fn format(&self, cursor_store: &mut CursorStore) -> String {
|
||||
let GrepFormatter {
|
||||
matches,
|
||||
files,
|
||||
total_matched,
|
||||
next_file_offset,
|
||||
output_mode,
|
||||
max_results,
|
||||
show_context,
|
||||
auto_expand_defs,
|
||||
picker,
|
||||
} = *self;
|
||||
|
||||
let items = if matches.len() > max_results {
|
||||
&matches[..max_results]
|
||||
} else {
|
||||
matches
|
||||
};
|
||||
|
||||
if output_mode == OutputMode::FilesWithMatches {
|
||||
return format_files_with_matches(
|
||||
items,
|
||||
files,
|
||||
next_file_offset,
|
||||
auto_expand_defs,
|
||||
cursor_store,
|
||||
picker,
|
||||
);
|
||||
}
|
||||
|
||||
if output_mode == OutputMode::Count {
|
||||
return format_count(items, files, next_file_offset, cursor_store, picker);
|
||||
}
|
||||
|
||||
// output_mode == usage
|
||||
let mut lines: Vec<String> = Vec::new();
|
||||
let unique_files = {
|
||||
let mut seen = std::collections::HashSet::new();
|
||||
for m in items {
|
||||
seen.insert(m.file_index);
|
||||
}
|
||||
seen.len()
|
||||
};
|
||||
|
||||
let max_output_chars: usize = if output_mode == OutputMode::Usage || unique_files <= 3 {
|
||||
5000
|
||||
} else if unique_files <= 8 {
|
||||
3500
|
||||
} else {
|
||||
2500
|
||||
};
|
||||
|
||||
// File overview: collect first match per file
|
||||
let file_preview = collect_file_preview(items, files, picker);
|
||||
let mut content_def_file = String::new();
|
||||
let mut content_first_file = String::new();
|
||||
for fm in &file_preview {
|
||||
if content_first_file.is_empty() {
|
||||
content_first_file = fm.file.relative_path(picker);
|
||||
}
|
||||
if content_def_file.is_empty() && fm.is_definition {
|
||||
content_def_file = fm.file.relative_path(picker);
|
||||
}
|
||||
}
|
||||
|
||||
let content_suggest = if !content_def_file.is_empty() {
|
||||
&content_def_file
|
||||
} else {
|
||||
&content_first_file
|
||||
};
|
||||
if !content_suggest.is_empty() {
|
||||
let file_count = file_preview.len();
|
||||
if file_count == 1 {
|
||||
lines.push(format!("→ Read {} (only match)", content_suggest));
|
||||
} else if !content_def_file.is_empty() {
|
||||
lines.push(format!("→ Read {} [def]", content_suggest));
|
||||
} else if file_count <= 3 {
|
||||
lines.push(format!("→ Read {} (best match)", content_suggest));
|
||||
}
|
||||
}
|
||||
|
||||
if total_matched > items.len() {
|
||||
lines.push(format!("{}/{} matches shown", items.len(), total_matched));
|
||||
}
|
||||
|
||||
// Track which files already had a definition expanded
|
||||
let mut def_expanded_files = std::collections::HashSet::new();
|
||||
|
||||
// Detailed content (subject to budget)
|
||||
let mut char_count = 0usize;
|
||||
let mut shown_count = 0usize;
|
||||
let mut current_file = String::new();
|
||||
|
||||
// Reorder: definitions first, then usages, then imports (when auto-expanding)
|
||||
let sorted_items: Vec<usize> = if auto_expand_defs {
|
||||
let mut indices: Vec<usize> = (0..items.len()).collect();
|
||||
indices.sort_unstable_by_key(|&i| {
|
||||
if items[i].is_definition {
|
||||
0
|
||||
} else if is_import_line(&items[i].line_content) {
|
||||
2
|
||||
} else {
|
||||
1
|
||||
}
|
||||
});
|
||||
|
||||
indices
|
||||
} else {
|
||||
(0..items.len()).collect()
|
||||
};
|
||||
|
||||
for &idx in &sorted_items {
|
||||
let m = &items[idx];
|
||||
let file = files[m.file_index];
|
||||
let mut match_lines: Vec<String> = Vec::new();
|
||||
|
||||
let file_rel_path = file.relative_path(picker);
|
||||
if file_rel_path != current_file {
|
||||
current_file = file_rel_path;
|
||||
match_lines.push(current_file.to_string());
|
||||
}
|
||||
|
||||
// Skip import-only lines when we already have definitions
|
||||
if auto_expand_defs && is_import_line(&m.line_content) && !def_expanded_files.is_empty()
|
||||
{
|
||||
continue;
|
||||
}
|
||||
|
||||
// Context before (only when explicitly requested)
|
||||
if show_context && !m.context_before.is_empty() {
|
||||
let start_line = m.line_number.saturating_sub(m.context_before.len() as u64);
|
||||
for (i, ctx) in m.context_before.iter().enumerate() {
|
||||
match_lines.push(format!(
|
||||
" {}-{}",
|
||||
start_line + i as u64,
|
||||
trauncate_line_for_ai(ctx, None, MAX_LINE_LEN)
|
||||
));
|
||||
}
|
||||
}
|
||||
|
||||
// Match line
|
||||
match_lines.push(format!(
|
||||
" {}: {}",
|
||||
m.line_number,
|
||||
trauncate_line_for_ai(
|
||||
&m.line_content,
|
||||
Some(m.match_byte_offsets.as_ref()),
|
||||
MAX_LINE_LEN
|
||||
)
|
||||
));
|
||||
|
||||
// Context after (only when explicitly requested via context parameter)
|
||||
if show_context && !m.context_after.is_empty() {
|
||||
let start_line = m.line_number + 1;
|
||||
for (i, ctx) in m.context_after.iter().enumerate() {
|
||||
match_lines.push(format!(
|
||||
" {}-{}",
|
||||
start_line + i as u64,
|
||||
trauncate_line_for_ai(ctx, None, MAX_LINE_LEN)
|
||||
));
|
||||
}
|
||||
match_lines.push("--".to_string());
|
||||
}
|
||||
|
||||
// Auto-expand definitions with body context
|
||||
let file_rel_for_expand = file.relative_path(picker);
|
||||
if auto_expand_defs
|
||||
&& !show_context
|
||||
&& m.is_definition
|
||||
&& !m.context_after.is_empty()
|
||||
&& !def_expanded_files.contains(&file_rel_for_expand)
|
||||
{
|
||||
let expand_limit = if def_expanded_files.is_empty() {
|
||||
MAX_DEF_EXPAND_FIRST
|
||||
} else {
|
||||
MAX_DEF_EXPAND
|
||||
};
|
||||
def_expanded_files.insert(file_rel_for_expand);
|
||||
let start_line = m.line_number + 1;
|
||||
for (i, ctx) in m.context_after.iter().take(expand_limit).enumerate() {
|
||||
if ctx.trim().is_empty() {
|
||||
break;
|
||||
}
|
||||
match_lines.push(format!(
|
||||
" {}| {}",
|
||||
start_line + i as u64,
|
||||
trauncate_line_for_ai(ctx, None, MAX_LINE_LEN)
|
||||
));
|
||||
}
|
||||
}
|
||||
|
||||
let chunk = match_lines.join("\n");
|
||||
if char_count + chunk.len() > max_output_chars && shown_count > 0 {
|
||||
break;
|
||||
}
|
||||
|
||||
char_count += chunk.len();
|
||||
lines.push(chunk);
|
||||
shown_count += 1;
|
||||
}
|
||||
|
||||
if next_file_offset > 0 {
|
||||
let cursor_id = cursor_store.store(next_file_offset);
|
||||
lines.push(format!("\ncursor: {}", cursor_id));
|
||||
}
|
||||
|
||||
lines.join("\n")
|
||||
}
|
||||
}
|
||||
|
||||
fn format_files_with_matches(
|
||||
items: &[GrepMatch],
|
||||
files: &[&FileItem],
|
||||
next_file_offset: usize,
|
||||
auto_expand_defs: bool,
|
||||
cursor_store: &mut CursorStore,
|
||||
picker: &FilePicker,
|
||||
) -> String {
|
||||
let file_map = collect_file_preview(items, files, picker);
|
||||
|
||||
let mut lines: Vec<String> = Vec::new();
|
||||
let file_count = file_map.len();
|
||||
|
||||
// Find best Read target
|
||||
let mut first_def_file = String::new();
|
||||
let mut first_file = String::new();
|
||||
for fm in &file_map {
|
||||
if first_file.is_empty() {
|
||||
first_file = fm.file.relative_path(picker);
|
||||
}
|
||||
if first_def_file.is_empty() && fm.is_definition {
|
||||
first_def_file = fm.file.relative_path(picker);
|
||||
}
|
||||
}
|
||||
let suggest_path = if !first_def_file.is_empty() {
|
||||
&first_def_file
|
||||
} else {
|
||||
&first_file
|
||||
};
|
||||
|
||||
if !suggest_path.is_empty() {
|
||||
if file_count == 1 {
|
||||
lines.push(format!(
|
||||
"→ Read {} (only match — no need to search further)",
|
||||
suggest_path
|
||||
));
|
||||
} else if !first_def_file.is_empty() && file_count <= 5 {
|
||||
lines.push(format!("→ Read {} (definition found)", suggest_path));
|
||||
} else if !first_def_file.is_empty() {
|
||||
lines.push(format!("→ Read {} (definition)", suggest_path));
|
||||
} else if file_count <= 3 {
|
||||
lines.push(format!("→ Read {} (best match)", suggest_path));
|
||||
} else {
|
||||
lines.push(format!("→ Read {}", suggest_path));
|
||||
}
|
||||
}
|
||||
|
||||
let is_small_set = file_count <= 5;
|
||||
let mut def_expanded_count = 0usize;
|
||||
|
||||
for (file_idx, fm) in file_map.iter().enumerate() {
|
||||
let is_def = fm.is_definition;
|
||||
let def_tag = if is_def { " [def]" } else { "" };
|
||||
lines.push(format!(
|
||||
"{}{}{}",
|
||||
fm.file.relative_path(picker),
|
||||
def_tag,
|
||||
size_tag(fm.file.size)
|
||||
));
|
||||
|
||||
// Show preview
|
||||
if !fm.line_content.is_empty() && (is_def || file_idx == 0 || is_small_set) {
|
||||
let ranges_ref: Option<&[(u32, u32)]> = if fm.match_ranges.is_empty() {
|
||||
None
|
||||
} else {
|
||||
Some(&fm.match_ranges)
|
||||
};
|
||||
lines.push(format!(
|
||||
" {}: {}",
|
||||
fm.line_number,
|
||||
trauncate_line_for_ai(&fm.line_content, ranges_ref, MAX_PREVIEW)
|
||||
));
|
||||
|
||||
// Auto-expand body context
|
||||
if auto_expand_defs && !fm.context_after.is_empty() {
|
||||
let expand_limit = if is_def {
|
||||
let limit = if def_expanded_count == 0 {
|
||||
MAX_DEF_EXPAND_FIRST
|
||||
} else {
|
||||
MAX_DEF_EXPAND
|
||||
};
|
||||
def_expanded_count += 1;
|
||||
limit
|
||||
} else if is_small_set && file_idx == 0 {
|
||||
MAX_FIRST_MATCH_EXPAND
|
||||
} else if is_small_set {
|
||||
MAX_DEF_EXPAND
|
||||
} else {
|
||||
0
|
||||
};
|
||||
|
||||
if expand_limit > 0 {
|
||||
let start_line = fm.line_number + 1;
|
||||
for (i, ctx) in fm.context_after.iter().take(expand_limit).enumerate() {
|
||||
if ctx.trim().is_empty() {
|
||||
break;
|
||||
}
|
||||
lines.push(format!(
|
||||
" {}| {}",
|
||||
start_line + i as u64,
|
||||
trauncate_line_for_ai(ctx, None, MAX_PREVIEW)
|
||||
));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if next_file_offset > 0 {
|
||||
let cursor_id = cursor_store.store(next_file_offset);
|
||||
lines.push(format!("\ncursor: {}", cursor_id));
|
||||
}
|
||||
|
||||
lines.join("\n")
|
||||
}
|
||||
|
||||
fn format_count(
|
||||
items: &[GrepMatch],
|
||||
files: &[&FileItem],
|
||||
next_file_offset: usize,
|
||||
cursor_store: &mut CursorStore,
|
||||
picker: &FilePicker,
|
||||
) -> String {
|
||||
let mut counts: std::collections::HashMap<String, usize> = std::collections::HashMap::new();
|
||||
let mut order: Vec<String> = Vec::new();
|
||||
for m in items {
|
||||
let file = files[m.file_index];
|
||||
let path = file.relative_path(picker);
|
||||
let count = counts.entry(path.to_string()).or_insert_with(|| {
|
||||
order.push(path.to_string());
|
||||
0
|
||||
});
|
||||
*count += 1;
|
||||
}
|
||||
|
||||
let mut lines: Vec<String> = Vec::new();
|
||||
for path in &order {
|
||||
lines.push(format!("{}: {}", path, counts[path.as_str()]));
|
||||
}
|
||||
if next_file_offset > 0 {
|
||||
let cursor_id = cursor_store.store(next_file_offset);
|
||||
lines.push(format!("\ncursor: {}", cursor_id));
|
||||
}
|
||||
lines.join("\n")
|
||||
}
|
||||
|
||||
fn collect_file_preview<'a>(
|
||||
items: &[GrepMatch],
|
||||
files: &[&'a FileItem],
|
||||
picker: &FilePicker,
|
||||
) -> Vec<FileMeta<'a>> {
|
||||
let mut file_preview: Vec<FileMeta<'a>> = Vec::new();
|
||||
let mut seen = std::collections::HashSet::new();
|
||||
for m in items {
|
||||
let file = files[m.file_index];
|
||||
if seen.insert(file.relative_path(picker)) {
|
||||
file_preview.push(FileMeta {
|
||||
file,
|
||||
line_number: m.line_number,
|
||||
line_content: m.line_content.clone(),
|
||||
is_definition: m.is_definition,
|
||||
match_ranges: m.match_byte_offsets.iter().copied().collect(),
|
||||
context_after: m.context_after.clone(),
|
||||
});
|
||||
}
|
||||
}
|
||||
file_preview
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn trunc_strips_trailing_whitespace() {
|
||||
// Leading whitespace is now stripped by core's trim_whitespace option.
|
||||
// This function only strips trailing whitespace.
|
||||
assert_eq!(trauncate_line_for_ai("foo()", None, 180), "foo()");
|
||||
assert_eq!(trauncate_line_for_ai("bar ", None, 180), "bar");
|
||||
assert_eq!(trauncate_line_for_ai(" ", None, 180), "");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn trunc_preserves_pre_trimmed_match_ranges() {
|
||||
// Core already stripped leading whitespace and adjusted offsets,
|
||||
// so "hello" arrives with match at bytes 0..5.
|
||||
let line = "hello";
|
||||
let ranges = [(0, 5)];
|
||||
let result = trauncate_line_for_ai(line, Some(&ranges), 180);
|
||||
assert_eq!(result, "hello");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn trunc_long_line_centered() {
|
||||
// Core already stripped leading whitespace; offsets are pre-adjusted.
|
||||
let line = format!("match_here{}", "x".repeat(200));
|
||||
let ranges = [(0u32, 10u32)];
|
||||
let result = trauncate_line_for_ai(&line, Some(&ranges), 50);
|
||||
assert!(result.contains("match_here"));
|
||||
assert!(result.len() <= 55); // budget + ellipsis chars
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,742 @@
|
||||
//! FFF MCP server — tool definitions and handlers.
|
||||
//!
|
||||
//! Uses the `rmcp` crate's `#[tool_router]` / `#[tool_handler]` macros
|
||||
//! for declarative tool registration. Each tool method directly calls
|
||||
//! `fff-core` APIs (no C FFI overhead).
|
||||
|
||||
use std::borrow::Cow;
|
||||
use std::sync::atomic::{AtomicBool, Ordering};
|
||||
use std::sync::{Arc, Mutex};
|
||||
|
||||
use crate::cursor::CursorStore;
|
||||
use crate::output::{GrepFormatter, OutputMode, file_suffix};
|
||||
use fff::grep::{GrepMode, GrepSearchOptions, has_regex_metacharacters};
|
||||
use fff::types::{FileItem, PaginationArgs};
|
||||
use fff::{FuzzySearchOptions, QueryParser, SharedFilePicker, SharedFrecency};
|
||||
use fff_query_parser::AiGrepConfig;
|
||||
use rmcp::handler::server::router::tool::ToolRouter;
|
||||
use rmcp::handler::server::wrapper::Parameters;
|
||||
use rmcp::model::*;
|
||||
use rmcp::{ServerHandler, schemars, tool, tool_handler, tool_router};
|
||||
|
||||
/// Normalize the caller-supplied `maxResults`.
|
||||
///
|
||||
/// `None`, `Some(0)`, and non-positive / non-finite values fall back to
|
||||
/// `default`. Issue #400 reported that grep returned 0 items for
|
||||
/// `maxResults: 0` while `find_files` returned the entire dataset; treating
|
||||
/// 0 as "use the default" makes both tools behave consistently.
|
||||
fn normalize_max_results(raw: Option<f64>, default: usize) -> usize {
|
||||
match raw {
|
||||
None => default,
|
||||
Some(v) if v <= 0.0 || !v.is_finite() => default,
|
||||
Some(v) => (v.round() as usize).max(1),
|
||||
}
|
||||
}
|
||||
|
||||
fn cleanup_fuzzy_query(s: &str) -> String {
|
||||
let mut out = String::with_capacity(s.len());
|
||||
for c in s.chars() {
|
||||
if !matches!(c, ':' | '-' | '_') {
|
||||
out.extend(c.to_lowercase());
|
||||
}
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
fn make_grep_options(
|
||||
output_mode: OutputMode,
|
||||
mode: GrepMode,
|
||||
file_offset: usize,
|
||||
context: Option<usize>,
|
||||
) -> (GrepSearchOptions, bool) {
|
||||
let is_usage = output_mode == OutputMode::Usage;
|
||||
let matches_per_file = match output_mode {
|
||||
OutputMode::FilesWithMatches => 1,
|
||||
_ if is_usage => 8,
|
||||
_ => 10,
|
||||
};
|
||||
let ctx_lines = if is_usage {
|
||||
context.unwrap_or(1)
|
||||
} else {
|
||||
context.unwrap_or(0)
|
||||
};
|
||||
let auto_expand = !is_usage && ctx_lines == 0;
|
||||
let after_ctx = if auto_expand { 8 } else { ctx_lines };
|
||||
|
||||
(
|
||||
GrepSearchOptions {
|
||||
max_file_size: 10 * 1024 * 1024,
|
||||
max_matches_per_file: matches_per_file,
|
||||
smart_case: true,
|
||||
file_offset,
|
||||
page_limit: 50,
|
||||
mode,
|
||||
time_budget_ms: 0,
|
||||
before_context: ctx_lines,
|
||||
after_context: after_ctx,
|
||||
classify_definitions: true,
|
||||
trim_whitespace: true,
|
||||
abort_signal: None,
|
||||
},
|
||||
auto_expand,
|
||||
)
|
||||
}
|
||||
|
||||
#[derive(Debug, serde::Deserialize, schemars::JsonSchema)]
|
||||
pub struct FindFilesParams {
|
||||
/// Fuzzy search query. Supports path prefixes and glob constraints.
|
||||
// `pattern` alias for consistency with grep's alias and the common
|
||||
// file-search parameter name (#311).
|
||||
#[serde(alias = "pattern")]
|
||||
pub query: String,
|
||||
/// Max results (default 20).
|
||||
#[serde(rename = "maxResults")]
|
||||
// this has to be float because llms are stupid
|
||||
pub max_results: Option<f64>,
|
||||
/// Cursor from previous result. Only use if previous results weren't sufficient.
|
||||
pub cursor: Option<String>,
|
||||
}
|
||||
|
||||
#[derive(Debug, serde::Deserialize, schemars::JsonSchema)]
|
||||
pub struct GrepParams {
|
||||
/// Search text or regex query with optional constraint prefixes.
|
||||
/// Matches within single lines only — use ONE specific term, not multiple words.
|
||||
// `pattern` alias: LLMs that have seen multi_grep (which uses `patterns`)
|
||||
// routinely call grep with `pattern`; accept it instead of erroring out
|
||||
// with an unhelpful "missing field `query`" (#311).
|
||||
#[serde(alias = "pattern")]
|
||||
pub query: String,
|
||||
/// Max matching lines (default 20).
|
||||
#[serde(rename = "maxResults")]
|
||||
pub max_results: Option<f64>, // this has to be float because llms are stupid
|
||||
/// Cursor from previous result. Only use if previous results weren't sufficient.
|
||||
pub cursor: Option<String>,
|
||||
/// Output format (default 'content').
|
||||
pub output_mode: Option<String>,
|
||||
}
|
||||
|
||||
fn deserialize_patterns<'de, D>(deserializer: D) -> Result<Vec<String>, D::Error>
|
||||
where
|
||||
D: serde::Deserializer<'de>,
|
||||
{
|
||||
use serde::de;
|
||||
|
||||
struct PatternsVisitor;
|
||||
|
||||
impl<'de> de::Visitor<'de> for PatternsVisitor {
|
||||
type Value = Vec<String>;
|
||||
|
||||
fn expecting(&self, formatter: &mut std::fmt::Formatter) -> std::fmt::Result {
|
||||
formatter.write_str("a string, an array of strings, or a stringified JSON array")
|
||||
}
|
||||
|
||||
fn visit_str<E: de::Error>(self, v: &str) -> Result<Self::Value, E> {
|
||||
// Try to parse as JSON array first
|
||||
if v.starts_with('[')
|
||||
&& let Ok(parsed) = serde_json::from_str::<Vec<String>>(v)
|
||||
{
|
||||
return Ok(parsed);
|
||||
}
|
||||
Ok(vec![v.to_string()])
|
||||
}
|
||||
|
||||
fn visit_string<E: de::Error>(self, v: String) -> Result<Self::Value, E> {
|
||||
if v.starts_with('[')
|
||||
&& let Ok(parsed) = serde_json::from_str::<Vec<String>>(&v)
|
||||
{
|
||||
return Ok(parsed);
|
||||
}
|
||||
Ok(vec![v])
|
||||
}
|
||||
|
||||
fn visit_seq<A: de::SeqAccess<'de>>(self, mut seq: A) -> Result<Self::Value, A::Error> {
|
||||
let mut values = Vec::new();
|
||||
while let Some(value) = seq.next_element::<String>()? {
|
||||
values.push(value);
|
||||
}
|
||||
Ok(values)
|
||||
}
|
||||
}
|
||||
|
||||
deserializer.deserialize_any(PatternsVisitor)
|
||||
}
|
||||
|
||||
#[derive(Debug, serde::Deserialize, schemars::JsonSchema)]
|
||||
pub struct MultiGrepParams {
|
||||
/// Patterns to match (OR logic). Include all naming conventions: snake_case, PascalCase, camelCase.
|
||||
#[serde(deserialize_with = "deserialize_patterns")]
|
||||
pub patterns: Vec<String>,
|
||||
/// File constraints (e.g. '*.{ts,tsx} !test/'). ALWAYS provide when possible.
|
||||
pub constraints: Option<String>,
|
||||
/// Max matching lines (default 20).
|
||||
#[serde(rename = "maxResults")]
|
||||
pub max_results: Option<f64>,
|
||||
/// Cursor from previous result.
|
||||
pub cursor: Option<String>,
|
||||
/// Output format (default 'content').
|
||||
pub output_mode: Option<String>,
|
||||
/// Context lines before/after each match.
|
||||
pub context: Option<f64>,
|
||||
}
|
||||
|
||||
#[derive(Clone)]
|
||||
pub struct FffServer {
|
||||
picker: SharedFilePicker,
|
||||
#[allow(dead_code)]
|
||||
frecency: SharedFrecency,
|
||||
cursor_store: Arc<Mutex<CursorStore>>,
|
||||
update_notice_sent: Arc<AtomicBool>,
|
||||
tool_router: ToolRouter<Self>,
|
||||
}
|
||||
|
||||
impl FffServer {
|
||||
pub fn new(picker: SharedFilePicker, frecency: SharedFrecency) -> Self {
|
||||
Self {
|
||||
picker,
|
||||
frecency,
|
||||
cursor_store: Arc::new(Mutex::new(CursorStore::new())),
|
||||
update_notice_sent: Arc::new(AtomicBool::new(false)),
|
||||
tool_router: Self::tool_router(),
|
||||
}
|
||||
}
|
||||
|
||||
#[allow(dead_code)]
|
||||
pub fn wait_for_scan(&self) {
|
||||
loop {
|
||||
let guard = self.picker.read().ok();
|
||||
let is_scanning = guard
|
||||
.as_ref()
|
||||
.and_then(|g| g.as_ref())
|
||||
.map(|p| p.is_scan_active())
|
||||
.unwrap_or(true);
|
||||
|
||||
if !is_scanning {
|
||||
break;
|
||||
}
|
||||
std::thread::sleep(std::time::Duration::from_millis(50));
|
||||
}
|
||||
}
|
||||
|
||||
fn lock_cursors(&self) -> Result<std::sync::MutexGuard<'_, CursorStore>, ErrorData> {
|
||||
self.cursor_store.lock().map_err(|e| {
|
||||
ErrorData::internal_error(format!("Failed to acquire cursor store lock: {e}"), None)
|
||||
})
|
||||
}
|
||||
|
||||
fn maybe_append_update_notice(&self, result: &mut CallToolResult) {
|
||||
if self.update_notice_sent.swap(true, Ordering::Relaxed) {
|
||||
return;
|
||||
}
|
||||
let notice = crate::update_check::get_update_notice();
|
||||
if notice.is_empty() {
|
||||
// Reset so the next call can try again (check may still be in flight)
|
||||
self.update_notice_sent.store(false, Ordering::Relaxed);
|
||||
return;
|
||||
}
|
||||
result.content.push(Content::text(notice));
|
||||
}
|
||||
|
||||
fn perform_grep(
|
||||
&self,
|
||||
query: &str,
|
||||
mode: GrepMode,
|
||||
max_results: usize,
|
||||
cursor_id: Option<&str>,
|
||||
output_mode: OutputMode,
|
||||
context: Option<usize>,
|
||||
) -> Result<CallToolResult, ErrorData> {
|
||||
let file_offset = cursor_id
|
||||
.and_then(|id| self.cursor_store.lock().ok()?.get(id))
|
||||
.unwrap_or(0);
|
||||
|
||||
let (options, auto_expand) = make_grep_options(output_mode, mode, file_offset, context);
|
||||
let ctx_lines = options.before_context;
|
||||
|
||||
// Acquire picker lock once for the entire operation.
|
||||
let guard = self.picker.read().map_err(|e| {
|
||||
ErrorData::internal_error(format!("Failed to acquire picker lock: {e}"), None)
|
||||
})?;
|
||||
let picker = guard
|
||||
.as_ref()
|
||||
.ok_or_else(|| ErrorData::internal_error("File picker not initialized", None))?;
|
||||
|
||||
let parser = QueryParser::new(AiGrepConfig);
|
||||
let parsed = parser.parse(query);
|
||||
let result = picker.grep(&parsed, &options);
|
||||
|
||||
if result.matches.is_empty() && file_offset == 0 {
|
||||
// Auto-retry: try broadening multi-word queries by dropping first non-constraint word
|
||||
let parts: Vec<&str> = query.split_whitespace().collect();
|
||||
if parts.len() >= 2 {
|
||||
let first_word = parts[0];
|
||||
let is_valid_constraint = first_word.starts_with('!')
|
||||
|| first_word.starts_with('*')
|
||||
|| first_word.ends_with('/');
|
||||
|
||||
if !is_valid_constraint {
|
||||
let rest_query = parts[1..].join(" ");
|
||||
let rest_parsed = parser.parse(&rest_query);
|
||||
|
||||
let rest_text = rest_parsed.grep_text();
|
||||
let retry_mode = if has_regex_metacharacters(&rest_text) {
|
||||
GrepMode::Regex
|
||||
} else {
|
||||
mode
|
||||
};
|
||||
|
||||
let (retry_options, _) = make_grep_options(output_mode, retry_mode, 0, context);
|
||||
let retry_result = picker.grep(&rest_parsed, &retry_options);
|
||||
|
||||
if !retry_result.matches.is_empty() && retry_result.matches.len() <= 10 {
|
||||
let mut cs = self.lock_cursors()?;
|
||||
let text = &GrepFormatter {
|
||||
matches: &retry_result.matches,
|
||||
files: &retry_result.files,
|
||||
total_matched: retry_result.matches.len(),
|
||||
next_file_offset: retry_result.next_file_offset,
|
||||
output_mode,
|
||||
max_results,
|
||||
show_context: ctx_lines > 0,
|
||||
auto_expand_defs: auto_expand,
|
||||
picker,
|
||||
}
|
||||
.format(&mut cs);
|
||||
return Ok(CallToolResult::success(vec![Content::text(format!(
|
||||
"0 matches for '{}'. Auto-broadened to '{}':\n{}",
|
||||
query, rest_query, text
|
||||
))]));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Fuzzy fallback for typo tolerance
|
||||
let fuzzy_query = cleanup_fuzzy_query(query);
|
||||
let (fuzzy_options, _) = make_grep_options(output_mode, GrepMode::Fuzzy, 0, Some(0));
|
||||
let fuzzy_parsed = parser.parse(&fuzzy_query);
|
||||
let fuzzy_result = picker.grep(&fuzzy_parsed, &fuzzy_options);
|
||||
|
||||
if !fuzzy_result.matches.is_empty() {
|
||||
let mut lines: Vec<String> = Vec::new();
|
||||
lines.push(format!(
|
||||
"0 exact matches. {} approximate:",
|
||||
fuzzy_result.matches.len()
|
||||
));
|
||||
let mut current_file = String::new();
|
||||
for m in fuzzy_result.matches.iter().take(3) {
|
||||
let file = fuzzy_result.files[m.file_index];
|
||||
let file_rel = file.relative_path(picker);
|
||||
if file_rel != current_file {
|
||||
current_file = file_rel;
|
||||
lines.push(current_file.to_string());
|
||||
}
|
||||
lines.push(format!(" {}: {}", m.line_number, m.line_content));
|
||||
}
|
||||
return Ok(CallToolResult::success(vec![Content::text(
|
||||
lines.join("\n"),
|
||||
)]));
|
||||
}
|
||||
|
||||
// File path fallback: if query looks like a path, suggest the matching file
|
||||
if query.contains('/') {
|
||||
let file_parser = QueryParser::default();
|
||||
let file_query = file_parser.parse(query);
|
||||
let file_opts = FuzzySearchOptions {
|
||||
max_threads: 0,
|
||||
current_file: None,
|
||||
project_path: Some(picker.base_path()),
|
||||
combo_boost_score_multiplier: 100,
|
||||
min_combo_count: 3,
|
||||
pagination: PaginationArgs {
|
||||
offset: 0,
|
||||
limit: 1,
|
||||
},
|
||||
};
|
||||
let file_result = picker.fuzzy_search(&file_query, None, file_opts);
|
||||
if let (Some(top), Some(score)) =
|
||||
(file_result.items.first(), file_result.scores.first())
|
||||
{
|
||||
// Only suggest when the match is strong enough.
|
||||
let query_len = query.len() as i32;
|
||||
if score.base_score > query_len * 10 {
|
||||
return Ok(CallToolResult::success(vec![Content::text(format!(
|
||||
"0 content matches. But there is a relevant file path: {}",
|
||||
top.relative_path(picker)
|
||||
))]));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return Ok(CallToolResult::success(vec![Content::text(
|
||||
"0 matches.".to_string(),
|
||||
)]));
|
||||
}
|
||||
|
||||
if result.matches.is_empty() {
|
||||
return Ok(CallToolResult::success(vec![Content::text(
|
||||
"0 matches.".to_string(),
|
||||
)]));
|
||||
}
|
||||
|
||||
let mut cs = self.lock_cursors()?;
|
||||
let text = &GrepFormatter {
|
||||
matches: &result.matches,
|
||||
files: &result.files,
|
||||
total_matched: result.matches.len(),
|
||||
next_file_offset: result.next_file_offset,
|
||||
output_mode,
|
||||
max_results,
|
||||
show_context: ctx_lines > 0,
|
||||
auto_expand_defs: auto_expand,
|
||||
picker,
|
||||
}
|
||||
.format(&mut cs);
|
||||
|
||||
Ok(CallToolResult::success(vec![Content::text(text)]))
|
||||
}
|
||||
}
|
||||
|
||||
#[tool_router]
|
||||
impl FffServer {
|
||||
/// Fuzzy file search by name. Searches FILE NAMES, not file contents.
|
||||
/// Use it when you need to find a file, not a definition.
|
||||
/// Use grep instead for searching code content (definitions, usage patterns).
|
||||
/// Supports fuzzy matching, path prefixes ('shc/'), and glob constraints.
|
||||
/// IMPORTANT: Keep queries SHORT — prefer 1-2 terms max.
|
||||
#[tool(
|
||||
name = "find_files",
|
||||
description = "Fuzzy file search by name. Searches FILE NAMES, not file contents. Use it when you need to find a file, not a definition. Use grep instead for searching code content (definitions, usage patterns). Supports fuzzy matching, path prefixes ('src/'), and glob constraints ('name **/src/*.{ts,tsx} !test/'). IMPORTANT: Keep queries SHORT — prefer 1-2 terms max. Multiple words are a waterfall (each narrows results), NOT OR. If unsure, start broad with 1 term and refine."
|
||||
)]
|
||||
fn find_files(
|
||||
&self,
|
||||
Parameters(params): Parameters<FindFilesParams>,
|
||||
) -> Result<CallToolResult, ErrorData> {
|
||||
let max_results = normalize_max_results(params.max_results, 20);
|
||||
let query = ¶ms.query;
|
||||
|
||||
let page_offset = params
|
||||
.cursor
|
||||
.as_deref()
|
||||
.and_then(|id| self.cursor_store.lock().ok()?.get(id))
|
||||
.unwrap_or(0);
|
||||
|
||||
let guard = self.picker.read().map_err(|e| {
|
||||
ErrorData::internal_error(format!("Failed to acquire picker lock: {e}"), None)
|
||||
})?;
|
||||
let picker = guard
|
||||
.as_ref()
|
||||
.ok_or_else(|| ErrorData::internal_error("File picker not initialized", None))?;
|
||||
let base_path = picker.base_path();
|
||||
let make_opts = |offset: usize| FuzzySearchOptions {
|
||||
max_threads: 0,
|
||||
current_file: None,
|
||||
project_path: Some(base_path),
|
||||
combo_boost_score_multiplier: 100,
|
||||
min_combo_count: 3,
|
||||
pagination: PaginationArgs {
|
||||
offset,
|
||||
limit: max_results,
|
||||
},
|
||||
};
|
||||
|
||||
let parser = QueryParser::default();
|
||||
let fff_query = parser.parse(query);
|
||||
let result = picker.fuzzy_search(&fff_query, None, make_opts(page_offset));
|
||||
let total_files = result.total_files;
|
||||
|
||||
// Auto-retry with fewer terms if 3+ words return 0 results
|
||||
let words: Vec<&str> = query.split_whitespace().collect();
|
||||
let shorter = words.get(..2).map(|w| w.join(" "));
|
||||
|
||||
let (items, scores, total_matched) =
|
||||
if result.items.is_empty() && words.len() >= 3 && page_offset == 0 {
|
||||
if let Some(shorter) = &shorter {
|
||||
let shorter_query = parser.parse(shorter);
|
||||
let retry = picker.fuzzy_search(&shorter_query, None, make_opts(0));
|
||||
|
||||
(retry.items, retry.scores, retry.total_matched)
|
||||
} else {
|
||||
(result.items, result.scores, result.total_matched)
|
||||
}
|
||||
} else {
|
||||
(result.items, result.scores, result.total_matched)
|
||||
};
|
||||
|
||||
if items.is_empty() {
|
||||
return Ok(CallToolResult::success(vec![Content::text(format!(
|
||||
"0 results ({} indexed)",
|
||||
total_files
|
||||
))]));
|
||||
}
|
||||
|
||||
let mut lines: Vec<String> = Vec::new();
|
||||
let top_item = items[0];
|
||||
let is_exact_match = scores[0].exact_match;
|
||||
|
||||
if page_offset == 0 {
|
||||
if is_exact_match {
|
||||
lines.push(format!(
|
||||
"→ Read {} (exact match!)",
|
||||
top_item.relative_path(picker)
|
||||
));
|
||||
} else if scores.len() < 2 || scores[0].total > scores[1].total.saturating_mul(2) {
|
||||
lines.push(format!(
|
||||
"→ Read {} (best match — Read this file directly)",
|
||||
top_item.relative_path(picker)
|
||||
));
|
||||
}
|
||||
}
|
||||
|
||||
let next_offset = page_offset + items.len();
|
||||
let has_more = next_offset < total_matched;
|
||||
|
||||
if has_more {
|
||||
lines.push(format!("{}/{} matches", items.len(), total_matched));
|
||||
}
|
||||
|
||||
for item in &items {
|
||||
lines.push(format!(
|
||||
"{}{}",
|
||||
item.relative_path(picker),
|
||||
file_suffix(item.git_status, item.total_frecency_score())
|
||||
));
|
||||
}
|
||||
|
||||
if has_more {
|
||||
let mut cs = self.lock_cursors()?;
|
||||
let cursor_id = cs.store(next_offset);
|
||||
lines.push(format!("cursor: {}", cursor_id));
|
||||
}
|
||||
|
||||
let mut result = CallToolResult::success(vec![Content::text(lines.join("\n"))]);
|
||||
self.maybe_append_update_notice(&mut result);
|
||||
Ok(result)
|
||||
}
|
||||
|
||||
/// Search file contents for text patterns. This is the DEFAULT search tool.
|
||||
/// Prefer plain text over regex. Filter files with constraints.
|
||||
#[tool(
|
||||
name = "grep",
|
||||
description = "Search file contents. Search for bare identifiers (e.g. 'InProgressQuote', 'ActorAuth'), NOT code syntax or regex. Filter files with constraints (e.g. '*.rs query', 'src/ query'). Use filename, directory (ending with /) or glob expressions to prefilter. See server instructions for constraint syntax and core rules."
|
||||
)]
|
||||
fn grep(
|
||||
&self,
|
||||
Parameters(params): Parameters<GrepParams>,
|
||||
) -> Result<CallToolResult, ErrorData> {
|
||||
let max_results = normalize_max_results(params.max_results, 20);
|
||||
let output_mode = OutputMode::new(params.output_mode.as_deref());
|
||||
|
||||
let parsed = QueryParser::new(AiGrepConfig).parse(¶ms.query);
|
||||
let grep_text = parsed.grep_text();
|
||||
|
||||
let mode = if has_regex_metacharacters(&grep_text) {
|
||||
GrepMode::Regex
|
||||
} else {
|
||||
GrepMode::PlainText
|
||||
};
|
||||
|
||||
let mut result = self.perform_grep(
|
||||
¶ms.query,
|
||||
mode,
|
||||
max_results,
|
||||
params.cursor.as_deref(),
|
||||
output_mode,
|
||||
None,
|
||||
)?;
|
||||
self.maybe_append_update_notice(&mut result);
|
||||
Ok(result)
|
||||
}
|
||||
|
||||
/// Search file contents for lines matching ANY of multiple patterns (OR logic).
|
||||
/// Patterns are literal text — NEVER escape special characters.
|
||||
#[tool(
|
||||
name = "multi_grep",
|
||||
description = "Search file contents for lines matching ANY of multiple patterns (OR logic). IMPORTANT: This returns files where ANY query matches, NOT all patterns. Patterns are literal text — NEVER escape special characters (no \\( \\) \\. etc). Faster than regex alternation for literal text. See server instructions for constraint syntax."
|
||||
)]
|
||||
fn multi_grep(
|
||||
&self,
|
||||
Parameters(params): Parameters<MultiGrepParams>,
|
||||
) -> Result<CallToolResult, ErrorData> {
|
||||
let mut result = self.multi_grep_inner(params)?;
|
||||
self.maybe_append_update_notice(&mut result);
|
||||
Ok(result)
|
||||
}
|
||||
}
|
||||
|
||||
impl FffServer {
|
||||
fn multi_grep_inner(&self, params: MultiGrepParams) -> Result<CallToolResult, ErrorData> {
|
||||
let max_results = normalize_max_results(params.max_results, 20);
|
||||
let context = params.context.map(|v| v.round() as usize);
|
||||
let output_mode = OutputMode::new(params.output_mode.as_deref());
|
||||
|
||||
let file_offset = params
|
||||
.cursor
|
||||
.as_deref()
|
||||
.and_then(|id| self.cursor_store.lock().ok()?.get(id))
|
||||
.unwrap_or(0);
|
||||
|
||||
let (options, auto_expand) =
|
||||
make_grep_options(output_mode, GrepMode::PlainText, file_offset, context);
|
||||
|
||||
let ctx_lines = options.before_context;
|
||||
let constraint_query = params.constraints.as_deref().unwrap_or("");
|
||||
let guard = self.picker.read().map_err(|e| {
|
||||
ErrorData::internal_error(format!("Failed to acquire picker lock: {e}"), None)
|
||||
})?;
|
||||
let picker = guard
|
||||
.as_ref()
|
||||
.ok_or_else(|| ErrorData::internal_error("File picker not initialized", None))?;
|
||||
let patterns_refs: Vec<&str> = params.patterns.iter().map(|s| s.as_str()).collect();
|
||||
|
||||
let parser = fff_query_parser::QueryParser::new(fff_query_parser::AiGrepConfig);
|
||||
let parsed_constraints = parser.parse(constraint_query);
|
||||
let constraints = parsed_constraints.constraints.as_slice();
|
||||
|
||||
let result = picker.multi_grep(&patterns_refs, constraints, &options);
|
||||
let file_refs: Vec<&FileItem> = result.files.to_vec();
|
||||
|
||||
if result.matches.is_empty() && file_offset == 0 {
|
||||
// Fallback: try individual patterns with plain grep
|
||||
let (fallback_options, _) =
|
||||
make_grep_options(output_mode, GrepMode::PlainText, 0, context);
|
||||
|
||||
let fallback_options = GrepSearchOptions {
|
||||
time_budget_ms: 3000,
|
||||
before_context: 0,
|
||||
..fallback_options
|
||||
};
|
||||
|
||||
for pat in ¶ms.patterns {
|
||||
let full_query: Cow<str> = if !constraint_query.is_empty() {
|
||||
Cow::Owned(format!("{} {}", constraint_query, pat))
|
||||
} else {
|
||||
Cow::Borrowed(pat)
|
||||
};
|
||||
|
||||
let parsed = parser.parse(&full_query);
|
||||
let fb_result = picker.grep(&parsed, &fallback_options);
|
||||
|
||||
if !fb_result.matches.is_empty() {
|
||||
let fb_file_refs: Vec<&FileItem> = fb_result.files.to_vec();
|
||||
let mut cs = self.lock_cursors()?;
|
||||
let text = &GrepFormatter {
|
||||
matches: &fb_result.matches,
|
||||
files: &fb_file_refs,
|
||||
total_matched: fb_result.matches.len(),
|
||||
next_file_offset: fb_result.next_file_offset,
|
||||
output_mode,
|
||||
max_results,
|
||||
show_context: false,
|
||||
auto_expand_defs: auto_expand,
|
||||
picker,
|
||||
}
|
||||
.format(&mut cs);
|
||||
return Ok(CallToolResult::success(vec![Content::text(format!(
|
||||
"0 multi-pattern matches. Plain grep fallback for \"{}\":\n{}",
|
||||
pat, text
|
||||
))]));
|
||||
}
|
||||
}
|
||||
|
||||
return Ok(CallToolResult::success(vec![Content::text(
|
||||
"0 matches.".to_string(),
|
||||
)]));
|
||||
}
|
||||
|
||||
if result.matches.is_empty() {
|
||||
return Ok(CallToolResult::success(vec![Content::text(
|
||||
"0 matches.".to_string(),
|
||||
)]));
|
||||
}
|
||||
|
||||
let mut cs = self.lock_cursors()?;
|
||||
let text = &GrepFormatter {
|
||||
matches: &result.matches,
|
||||
files: &file_refs,
|
||||
total_matched: result.matches.len(),
|
||||
next_file_offset: result.next_file_offset,
|
||||
output_mode,
|
||||
max_results,
|
||||
show_context: ctx_lines > 0,
|
||||
auto_expand_defs: auto_expand,
|
||||
picker,
|
||||
}
|
||||
.format(&mut cs);
|
||||
|
||||
Ok(CallToolResult::success(vec![Content::text(text)]))
|
||||
}
|
||||
}
|
||||
|
||||
#[tool_handler]
|
||||
impl ServerHandler for FffServer {
|
||||
fn get_info(&self) -> ServerInfo {
|
||||
let notice = crate::update_check::get_update_notice();
|
||||
let instructions = if notice.is_empty() {
|
||||
crate::MCP_INSTRUCTIONS.to_string()
|
||||
} else {
|
||||
format!("{}{}", crate::MCP_INSTRUCTIONS, notice)
|
||||
};
|
||||
|
||||
ServerInfo::new(ServerCapabilities::builder().enable_tools().build())
|
||||
.with_server_info(Implementation::new("fff", env!("CARGO_PKG_VERSION")))
|
||||
.with_instructions(instructions)
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn normalize_max_results_none_uses_default() {
|
||||
assert_eq!(normalize_max_results(None, 20), 20);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn normalize_max_results_zero_uses_default() {
|
||||
// Issue #400: `maxResults: 0` must not return zero items for grep
|
||||
// while `find_files` returns the full set. Both tools now map 0 to
|
||||
// the default limit.
|
||||
assert_eq!(normalize_max_results(Some(0.0), 20), 20);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn normalize_max_results_negative_uses_default() {
|
||||
assert_eq!(normalize_max_results(Some(-5.0), 20), 20);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn normalize_max_results_non_finite_uses_default() {
|
||||
assert_eq!(normalize_max_results(Some(f64::NAN), 20), 20);
|
||||
assert_eq!(normalize_max_results(Some(f64::INFINITY), 20), 20);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn normalize_max_results_rounds_and_clamps() {
|
||||
assert_eq!(normalize_max_results(Some(0.4), 20), 1);
|
||||
assert_eq!(normalize_max_results(Some(10.0), 20), 10);
|
||||
assert_eq!(normalize_max_results(Some(10.7), 20), 11);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn grep_params_accepts_pattern_alias() {
|
||||
// Issue #311: LLMs flip between `query` and `pattern`; accept both.
|
||||
let via_query: GrepParams =
|
||||
serde_json::from_str(r#"{"query":"foo"}"#).expect("query field");
|
||||
assert_eq!(via_query.query, "foo");
|
||||
|
||||
let via_pattern: GrepParams =
|
||||
serde_json::from_str(r#"{"pattern":"foo"}"#).expect("pattern alias");
|
||||
assert_eq!(via_pattern.query, "foo");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn find_files_params_accepts_pattern_alias() {
|
||||
let via_query: FindFilesParams =
|
||||
serde_json::from_str(r#"{"query":"foo"}"#).expect("query field");
|
||||
assert_eq!(via_query.query, "foo");
|
||||
|
||||
let via_pattern: FindFilesParams =
|
||||
serde_json::from_str(r#"{"pattern":"foo"}"#).expect("pattern alias");
|
||||
assert_eq!(via_pattern.query, "foo");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,78 @@
|
||||
//! Background update checker — compares the embedded build hash against
|
||||
//! the latest GitHub release tag to surface upgrade notices in MCP instructions.
|
||||
|
||||
use std::sync::OnceLock;
|
||||
|
||||
const REPO: &str = "dmtrKovalenko/fff.nvim";
|
||||
const BUILD_HASH: &str = env!("FFF_GIT_HASH");
|
||||
|
||||
/// Holds the result of the update check (empty string = up to date or check failed).
|
||||
static UPDATE_NOTICE: OnceLock<String> = OnceLock::new();
|
||||
|
||||
/// Returns the update notice if the check has completed, empty string otherwise.
|
||||
pub fn get_update_notice() -> &'static str {
|
||||
UPDATE_NOTICE.get().map(|s| s.as_str()).unwrap_or("")
|
||||
}
|
||||
|
||||
/// Kick off the update check in a background thread so it never blocks the server.
|
||||
pub fn spawn_update_check() {
|
||||
std::thread::spawn(|| {
|
||||
let notice = check_latest_release();
|
||||
let _ = UPDATE_NOTICE.set(notice);
|
||||
});
|
||||
}
|
||||
|
||||
/// Fetch the latest release tag from GitHub and compare against the build hash.
|
||||
fn check_latest_release() -> String {
|
||||
match fetch_latest_tag() {
|
||||
Ok(tag) => compare_versions(BUILD_HASH, &tag),
|
||||
Err(_) => String::new(),
|
||||
}
|
||||
}
|
||||
|
||||
/// Compare a build hash against a release tag.
|
||||
/// Returns an update notice string, or empty if up-to-date.
|
||||
fn compare_versions(build_hash: &str, release_tag: &str) -> String {
|
||||
let tag = release_tag.trim();
|
||||
if tag.is_empty() || build_hash == "unknown" {
|
||||
return String::new();
|
||||
}
|
||||
|
||||
let our_short = &build_hash[..build_hash.len().min(tag.len())];
|
||||
if our_short == tag {
|
||||
return String::new();
|
||||
}
|
||||
|
||||
format!(
|
||||
"\n[fff update available: `curl -fsSL https://raw.githubusercontent.com/{REPO}/main/install-mcp.sh | bash`]\n"
|
||||
)
|
||||
}
|
||||
|
||||
/// Shell out to curl to fetch the latest release tag name from GitHub API.
|
||||
fn fetch_latest_tag() -> Result<String, Box<dyn std::error::Error>> {
|
||||
let output = std::process::Command::new("curl")
|
||||
.args([
|
||||
"-fsSL",
|
||||
"--max-time",
|
||||
"5",
|
||||
"-H",
|
||||
"Accept: application/vnd.github.v3+json",
|
||||
&format!("https://api.github.com/repos/{REPO}/releases?per_page=1"),
|
||||
])
|
||||
.output()?;
|
||||
|
||||
if !output.status.success() {
|
||||
return Err("curl failed".into());
|
||||
}
|
||||
|
||||
let body = String::from_utf8(output.stdout)?;
|
||||
let releases: Vec<serde_json::Value> = serde_json::from_str(&body)?;
|
||||
let tag = releases
|
||||
.first()
|
||||
.and_then(|r| r.get("tag_name"))
|
||||
.and_then(|v| v.as_str())
|
||||
.unwrap_or("")
|
||||
.to_string();
|
||||
|
||||
Ok(tag)
|
||||
}
|
||||
+43
-29
@@ -1,12 +1,16 @@
|
||||
[package]
|
||||
name = "fff-nvim"
|
||||
version = "0.1.0"
|
||||
version = "0.8.1"
|
||||
edition = "2024"
|
||||
|
||||
[lib]
|
||||
path = "src/lib.rs"
|
||||
crate-type = ["cdylib", "rlib"]
|
||||
|
||||
[features]
|
||||
default = []
|
||||
zlob = ["fff/zlob"]
|
||||
|
||||
[[bin]]
|
||||
name = "test_watcher"
|
||||
path = "src/bin/test_watcher.rs"
|
||||
@@ -31,54 +35,64 @@ path = "src/bin/grep_profiler.rs"
|
||||
name = "grep_vs_rg"
|
||||
path = "src/bin/grep_vs_rg.rs"
|
||||
|
||||
[[bin]]
|
||||
name = "bench_grep_query"
|
||||
path = "src/bin/bench_grep_query.rs"
|
||||
|
||||
[[bin]]
|
||||
name = "fuzzy_grep_test"
|
||||
path = "src/bin/fuzzy_grep_test.rs"
|
||||
|
||||
[[bin]]
|
||||
name = "test_memory_leak"
|
||||
path = "src/bin/test_memory_leak.rs"
|
||||
|
||||
[[bin]]
|
||||
name = "bench_ci_memmem"
|
||||
path = "src/bin/bench_ci_memmem.rs"
|
||||
|
||||
[[bin]]
|
||||
name = "bench_lmdb_parallel"
|
||||
path = "src/bin/bench_lmdb_parallel.rs"
|
||||
|
||||
[dependencies]
|
||||
# Workspace dependencies
|
||||
ahash = { workspace = true }
|
||||
rayon = { workspace = true }
|
||||
smallvec = { workspace = true }
|
||||
thiserror = { workspace = true }
|
||||
tracing = { workspace = true }
|
||||
|
||||
# Local crates
|
||||
fff-core = { path = "../fff-core" }
|
||||
fff-query-parser = { path = "../fff-query-parser" }
|
||||
|
||||
# External dependencies
|
||||
blake3 = "1.8.2"
|
||||
fff = { package = "fff-search", path = "../fff-core", version = "0.8.1", features = [
|
||||
"mimalloc-collect",
|
||||
] }
|
||||
fff-query-parser = { path = "../fff-query-parser", version = "0.8.1" }
|
||||
chrono = { version = "0.4", features = ["serde"] }
|
||||
ctrlc = "3.4.2"
|
||||
dirs = "5.0"
|
||||
git2 = { workspace = true }
|
||||
glidesort = "0.1"
|
||||
heed = "0.22.0"
|
||||
ignore = "0.4.22"
|
||||
mimalloc = "0.1.47"
|
||||
mimalloc = { version = "0.1.47", features = ["local_dynamic_tls"] }
|
||||
mlua = { version = "0.11.1", features = ["module", "luajit"] }
|
||||
neo_frizbee = { workspace = true }
|
||||
notify = "8.1.0"
|
||||
notify-debouncer-full = "0.6"
|
||||
once_cell = "1.20.2"
|
||||
pathdiff = "0.2.1"
|
||||
serde = { version = "1.0", features = ["derive"] }
|
||||
smartstring = { version = "1.0.1", features = ["serde"] }
|
||||
tracing-appender = "0.2"
|
||||
tracing-subscriber = { version = "0.3", features = ["env-filter"] }
|
||||
zlob = { workspace = true }
|
||||
|
||||
[dev-dependencies]
|
||||
criterion = { version = "0.5", features = ["html_reports"] }
|
||||
rand = { version = "0.8", features = ["small_rng"] }
|
||||
tempfile = "3.8"
|
||||
|
||||
[[bench]]
|
||||
name = "indexing_and_search"
|
||||
name = "fuzzy_search"
|
||||
path = "benches/fuzzy_search_bench.rs"
|
||||
harness = false
|
||||
|
||||
[[bench]]
|
||||
name = "query_tracker_bench"
|
||||
name = "grep_bench"
|
||||
path = "benches/grep_bench.rs"
|
||||
harness = false
|
||||
|
||||
# Platform-specific: Use vendored OpenSSL on non-Windows (Linux, macOS)
|
||||
# On Windows, git2 uses the native SChannel TLS backend
|
||||
[target.'cfg(not(windows))'.dependencies]
|
||||
openssl = { version = "0.10", features = ["vendored"] }
|
||||
[[bench]]
|
||||
name = "query_tracker"
|
||||
path = "benches/query_tracker_bench.rs"
|
||||
harness = false
|
||||
|
||||
[[bench]]
|
||||
name = "scan"
|
||||
path = "benches/scan_bench.rs"
|
||||
harness = false
|
||||
|
||||
+212
-203
@@ -1,7 +1,10 @@
|
||||
use criterion::{BenchmarkId, Criterion, black_box, criterion_group, criterion_main};
|
||||
use fff_nvim::FILE_PICKER;
|
||||
use fff_nvim::file_picker::{FilePicker, FuzzySearchOptions};
|
||||
use fff_nvim::types::PaginationArgs;
|
||||
use fff::file_picker::{FFFMode, FilePicker};
|
||||
use fff::types::PaginationArgs;
|
||||
use fff::{
|
||||
FilePickerOptions, FuzzySearchOptions, GrepMode, GrepSearchOptions, QueryParser,
|
||||
SharedFilePicker, SharedFrecency,
|
||||
};
|
||||
use std::path::PathBuf;
|
||||
use std::time::Duration;
|
||||
|
||||
@@ -19,20 +22,30 @@ fn init_tracing() {
|
||||
// .try_init();
|
||||
}
|
||||
|
||||
/// Initialize FilePicker and insert into global state
|
||||
fn init_file_picker_internal(path: &str) -> Result<(), String> {
|
||||
let picker = FilePicker::new(path.to_string())
|
||||
.map_err(|e| format!("Failed to create FilePicker: {:?}", e))?;
|
||||
|
||||
let mut picker_guard = FILE_PICKER
|
||||
.write()
|
||||
.map_err(|_| "Failed to acquire write lock")?;
|
||||
*picker_guard = Some(picker);
|
||||
Ok(())
|
||||
/// Initialize FilePicker using shared state
|
||||
fn init_file_picker_internal(
|
||||
path: &str,
|
||||
shared_picker: &SharedFilePicker,
|
||||
shared_frecency: &SharedFrecency,
|
||||
) -> Result<(), String> {
|
||||
FilePicker::new_with_shared_state(
|
||||
shared_picker.clone(),
|
||||
shared_frecency.clone(),
|
||||
FilePickerOptions {
|
||||
base_path: path.to_string(),
|
||||
enable_mmap_cache: false,
|
||||
mode: FFFMode::Neovim,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.map_err(|e| format!("Failed to create FilePicker: {:?}", e))
|
||||
}
|
||||
|
||||
/// Helper function to wait for scanning to complete and get file count
|
||||
fn wait_for_scan_completion(timeout_secs: u64) -> Result<usize, String> {
|
||||
fn wait_for_scan_completion(
|
||||
shared_picker: &SharedFilePicker,
|
||||
timeout_secs: u64,
|
||||
) -> Result<usize, String> {
|
||||
let start = std::time::Instant::now();
|
||||
let timeout = Duration::from_secs(timeout_secs);
|
||||
let mut last_log = std::time::Instant::now();
|
||||
@@ -42,7 +55,7 @@ fn wait_for_scan_completion(timeout_secs: u64) -> Result<usize, String> {
|
||||
iteration += 1;
|
||||
|
||||
{
|
||||
let picker_guard = FILE_PICKER
|
||||
let picker_guard = shared_picker
|
||||
.read()
|
||||
.map_err(|_| "Failed to acquire read lock")?;
|
||||
if let Some(ref picker) = *picker_guard {
|
||||
@@ -91,29 +104,17 @@ fn wait_for_scan_completion(timeout_secs: u64) -> Result<usize, String> {
|
||||
}
|
||||
}
|
||||
|
||||
/// Get files from the global FILE_PICKER
|
||||
fn get_files_snapshot() -> Result<Vec<fff_nvim::types::FileItem>, String> {
|
||||
let picker_guard = FILE_PICKER
|
||||
.read()
|
||||
.map_err(|_| "Failed to acquire read lock")?;
|
||||
if let Some(ref picker) = *picker_guard {
|
||||
Ok(picker.get_files().to_vec())
|
||||
} else {
|
||||
Err("FilePicker not initialized".to_string())
|
||||
}
|
||||
}
|
||||
|
||||
/// Clean up global state
|
||||
fn cleanup_global_state() {
|
||||
if let Ok(mut picker_guard) = FILE_PICKER.write() {
|
||||
/// Clean up shared state
|
||||
fn cleanup_shared_state(shared_picker: &SharedFilePicker) {
|
||||
if let Ok(mut picker_guard) = shared_picker.write() {
|
||||
if let Some(mut picker) = picker_guard.take() {
|
||||
picker.stop_background_monitor();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Initialize FilePicker once and return files snapshot
|
||||
fn setup_once() -> Result<Vec<fff_nvim::types::FileItem>, String> {
|
||||
/// Initialize FilePicker once and return shared state
|
||||
fn setup_once() -> Result<(SharedFilePicker, SharedFrecency), String> {
|
||||
init_tracing();
|
||||
|
||||
let big_repo_path = PathBuf::from("./big-repo");
|
||||
@@ -121,98 +122,42 @@ fn setup_once() -> Result<Vec<fff_nvim::types::FileItem>, String> {
|
||||
return Err("./big-repo directory does not exist. Run git clone https://github.com/torvalds/linux.git big-repo".to_string());
|
||||
}
|
||||
|
||||
let canonical_path = fff_core::path_utils::canonicalize(&big_repo_path)
|
||||
let canonical_path = fff::path_utils::canonicalize(&big_repo_path)
|
||||
.map_err(|e| format!("Failed to canonicalize path: {}", e))?;
|
||||
eprintln!(" Path: {:?}", canonical_path);
|
||||
|
||||
{
|
||||
let picker_guard = FILE_PICKER
|
||||
.read()
|
||||
.map_err(|_| "Failed to acquire read lock")?;
|
||||
if let Some(ref picker) = *picker_guard {
|
||||
let files = picker.get_files();
|
||||
if !files.is_empty() {
|
||||
eprintln!(" ℹ Reusing existing index with {} files", files.len());
|
||||
return Ok(files.to_vec());
|
||||
}
|
||||
}
|
||||
}
|
||||
let shared_picker = SharedFilePicker::default();
|
||||
let shared_frecency = SharedFrecency::default();
|
||||
|
||||
cleanup_global_state();
|
||||
std::thread::sleep(Duration::from_millis(500));
|
||||
|
||||
init_file_picker_internal(&canonical_path.to_string_lossy())?;
|
||||
init_file_picker_internal(
|
||||
&canonical_path.to_string_lossy(),
|
||||
&shared_picker,
|
||||
&shared_frecency,
|
||||
)?;
|
||||
|
||||
eprintln!(" Waiting for background scan to complete...");
|
||||
let file_count = wait_for_scan_completion(120)?;
|
||||
let file_count = wait_for_scan_completion(&shared_picker, 120)?;
|
||||
eprintln!(
|
||||
" ✓ Indexed {} files (will be reused for all benchmarks)\n",
|
||||
file_count
|
||||
);
|
||||
|
||||
get_files_snapshot()
|
||||
}
|
||||
|
||||
/// Benchmark for indexing the big-repo directory
|
||||
fn bench_indexing(c: &mut Criterion) {
|
||||
init_tracing();
|
||||
|
||||
let big_repo_path = PathBuf::from("./big-repo");
|
||||
if !big_repo_path.exists() {
|
||||
eprintln!(
|
||||
"./big-repo directory does not exist. Run git clone https://github.com/torvalds/linux.git big-repo"
|
||||
);
|
||||
return;
|
||||
}
|
||||
|
||||
let canonical_path = match fff_core::path_utils::canonicalize(&big_repo_path) {
|
||||
Ok(p) => p,
|
||||
Err(e) => {
|
||||
eprintln!("⚠ Failed to canonicalize path: {}", e);
|
||||
return;
|
||||
}
|
||||
};
|
||||
|
||||
let mut group = c.benchmark_group("indexing");
|
||||
group.sample_size(10);
|
||||
group.measurement_time(Duration::from_secs(20));
|
||||
|
||||
group.bench_function("index_big_repo", |b| {
|
||||
b.iter(|| {
|
||||
cleanup_global_state();
|
||||
std::thread::sleep(Duration::from_millis(500));
|
||||
|
||||
let start = std::time::Instant::now();
|
||||
init_file_picker_internal(black_box(&canonical_path.to_string_lossy()))
|
||||
.expect("Failed to init FilePicker");
|
||||
|
||||
match wait_for_scan_completion(120) {
|
||||
Ok(file_count) => {
|
||||
let elapsed = start.elapsed();
|
||||
eprintln!(" ✓ Indexed {} files in {:?}", file_count, elapsed);
|
||||
file_count
|
||||
}
|
||||
Err(e) => {
|
||||
eprintln!(" ✗ Error: {}", e);
|
||||
0
|
||||
}
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
group.finish();
|
||||
Ok((shared_picker, shared_frecency))
|
||||
}
|
||||
|
||||
/// Benchmark for searching with various query patterns
|
||||
fn bench_search_queries(c: &mut Criterion) {
|
||||
let files = match setup_once() {
|
||||
Ok(files) => files,
|
||||
let (sp, _sf) = match setup_once() {
|
||||
Ok(result) => result,
|
||||
Err(e) => {
|
||||
eprint!("Failed to setup picker {e:?}");
|
||||
return;
|
||||
}
|
||||
};
|
||||
|
||||
let guard = sp.read().unwrap();
|
||||
let picker = guard.as_ref().unwrap();
|
||||
|
||||
let mut group = c.benchmark_group("search");
|
||||
group.sample_size(100);
|
||||
|
||||
@@ -224,18 +169,20 @@ fn bench_search_queries(c: &mut Criterion) {
|
||||
("partial", "src/lib"),
|
||||
];
|
||||
|
||||
let parser = QueryParser::default();
|
||||
|
||||
for (name, query) in test_queries {
|
||||
group.bench_with_input(BenchmarkId::new("query", name), &query, |b, &query| {
|
||||
let parsed = parser.parse(query);
|
||||
group.bench_with_input(BenchmarkId::new("query", name), &query, |b, &_query| {
|
||||
b.iter(|| {
|
||||
let results = FilePicker::fuzzy_search(
|
||||
black_box(&files),
|
||||
black_box(query),
|
||||
let results = picker.fuzzy_search(
|
||||
black_box(&parsed),
|
||||
None,
|
||||
FuzzySearchOptions {
|
||||
max_threads: 4,
|
||||
current_file: None,
|
||||
|
||||
project_path: None,
|
||||
last_same_query_match: None,
|
||||
|
||||
combo_boost_score_multiplier: 100,
|
||||
min_combo_count: 3,
|
||||
pagination: PaginationArgs {
|
||||
@@ -254,18 +201,23 @@ fn bench_search_queries(c: &mut Criterion) {
|
||||
|
||||
/// Benchmark search with different thread counts
|
||||
fn bench_search_thread_scaling(c: &mut Criterion) {
|
||||
let files = match setup_once() {
|
||||
Ok(files) => files,
|
||||
let (sp, _sf) = match setup_once() {
|
||||
Ok(result) => result,
|
||||
Err(e) => {
|
||||
eprintln!("⚠ Skipping thread scaling benchmarks: {}", e);
|
||||
eprintln!("Skipping thread scaling benchmarks: {}", e);
|
||||
return;
|
||||
}
|
||||
};
|
||||
|
||||
let guard = sp.read().unwrap();
|
||||
let picker = guard.as_ref().unwrap();
|
||||
|
||||
let mut group = c.benchmark_group("thread_scaling");
|
||||
group.sample_size(100);
|
||||
|
||||
let query = "controller";
|
||||
let parser = QueryParser::default();
|
||||
let parsed = parser.parse(query);
|
||||
let thread_counts = vec![1, 2, 4, 8];
|
||||
|
||||
for threads in thread_counts {
|
||||
@@ -274,15 +226,14 @@ fn bench_search_thread_scaling(c: &mut Criterion) {
|
||||
&threads,
|
||||
|b, &threads| {
|
||||
b.iter(|| {
|
||||
let results = FilePicker::fuzzy_search(
|
||||
black_box(&files),
|
||||
black_box(query),
|
||||
let results = picker.fuzzy_search(
|
||||
black_box(&parsed),
|
||||
None,
|
||||
FuzzySearchOptions {
|
||||
max_threads: threads,
|
||||
current_file: None,
|
||||
|
||||
project_path: None,
|
||||
last_same_query_match: None,
|
||||
|
||||
combo_boost_score_multiplier: 100,
|
||||
min_combo_count: 3,
|
||||
pagination: PaginationArgs {
|
||||
@@ -302,38 +253,39 @@ fn bench_search_thread_scaling(c: &mut Criterion) {
|
||||
|
||||
/// Benchmark search with different result limits
|
||||
fn bench_search_result_limits(c: &mut Criterion) {
|
||||
let files = match setup_once() {
|
||||
Ok(files) => files,
|
||||
let (sp, _sf) = match setup_once() {
|
||||
Ok(result) => result,
|
||||
Err(e) => {
|
||||
eprintln!("⚠ Skipping result limit benchmarks: {}", e);
|
||||
eprintln!("Skipping result limit benchmarks: {}", e);
|
||||
return;
|
||||
}
|
||||
};
|
||||
|
||||
let guard = sp.read().unwrap();
|
||||
let picker = guard.as_ref().unwrap();
|
||||
|
||||
let mut group = c.benchmark_group("result_limits");
|
||||
group.sample_size(100);
|
||||
|
||||
let query = "mod";
|
||||
let parser = QueryParser::default();
|
||||
let parsed = parser.parse(query);
|
||||
let result_limits = vec![10, 50, 100, 500];
|
||||
|
||||
for limit in result_limits {
|
||||
group.bench_with_input(BenchmarkId::from_parameter(limit), &limit, |b, &limit| {
|
||||
b.iter(|| {
|
||||
let results = FilePicker::fuzzy_search(
|
||||
black_box(&files),
|
||||
black_box(query),
|
||||
let results = picker.fuzzy_search(
|
||||
black_box(&parsed),
|
||||
None,
|
||||
FuzzySearchOptions {
|
||||
max_threads: 4,
|
||||
current_file: None,
|
||||
|
||||
project_path: None,
|
||||
last_same_query_match: None,
|
||||
|
||||
combo_boost_score_multiplier: 100,
|
||||
min_combo_count: 3,
|
||||
pagination: PaginationArgs {
|
||||
offset: 0,
|
||||
limit: limit,
|
||||
},
|
||||
pagination: PaginationArgs { offset: 0, limit },
|
||||
},
|
||||
);
|
||||
results.total_matched
|
||||
@@ -344,20 +296,23 @@ fn bench_search_result_limits(c: &mut Criterion) {
|
||||
group.finish();
|
||||
}
|
||||
|
||||
/// Benchmark search algorithm performance scaling with file count
|
||||
/// Benchmark search algorithm performance with queries of varying selectivity
|
||||
fn bench_search_scalability(c: &mut Criterion) {
|
||||
let all_files = match setup_once() {
|
||||
Ok(files) => files,
|
||||
let (sp, _sf) = match setup_once() {
|
||||
Ok(result) => result,
|
||||
Err(e) => {
|
||||
eprintln!("⚠ Skipping scalability benchmarks: {}", e);
|
||||
eprintln!("Skipping scalability benchmarks: {}", e);
|
||||
return;
|
||||
}
|
||||
};
|
||||
|
||||
if all_files.len() < 1000 {
|
||||
let guard = sp.read().unwrap();
|
||||
let picker = guard.as_ref().unwrap();
|
||||
|
||||
if picker.get_files().len() < 1000 {
|
||||
eprintln!(
|
||||
"⚠ Skipping scalability benchmark: need at least 1000 files, got {}",
|
||||
all_files.len()
|
||||
"Skipping scalability benchmark: need at least 1000 files, got {}",
|
||||
picker.get_files().len()
|
||||
);
|
||||
return;
|
||||
}
|
||||
@@ -365,26 +320,26 @@ fn bench_search_scalability(c: &mut Criterion) {
|
||||
let mut group = c.benchmark_group("search_scalability");
|
||||
group.sample_size(50);
|
||||
|
||||
let query = "controller";
|
||||
let file_counts = vec![100, 1000, 5000, 10000, all_files.len().min(50000)];
|
||||
let parser = QueryParser::default();
|
||||
let selectivity_queries = vec![
|
||||
("broad_a", "a"),
|
||||
("medium_mod", "mod"),
|
||||
("narrow_controller", "controller"),
|
||||
("very_narrow_user_auth", "user_authentication"),
|
||||
];
|
||||
|
||||
for count in file_counts {
|
||||
if count > all_files.len() {
|
||||
continue;
|
||||
}
|
||||
|
||||
let subset = &all_files[..count];
|
||||
group.bench_with_input(BenchmarkId::from_parameter(count), &count, |b, _| {
|
||||
for (name, query) in selectivity_queries {
|
||||
let parsed = parser.parse(query);
|
||||
group.bench_with_input(BenchmarkId::from_parameter(name), &name, |b, _| {
|
||||
b.iter(|| {
|
||||
let results = FilePicker::fuzzy_search(
|
||||
black_box(subset),
|
||||
black_box(query),
|
||||
let results = picker.fuzzy_search(
|
||||
black_box(&parsed),
|
||||
None,
|
||||
FuzzySearchOptions {
|
||||
max_threads: 4,
|
||||
current_file: None,
|
||||
|
||||
project_path: None,
|
||||
last_same_query_match: None,
|
||||
|
||||
combo_boost_score_multiplier: 100,
|
||||
min_combo_count: 3,
|
||||
pagination: PaginationArgs {
|
||||
@@ -403,31 +358,35 @@ fn bench_search_scalability(c: &mut Criterion) {
|
||||
|
||||
/// Benchmark search performance with different ordering modes
|
||||
fn bench_search_ordering(c: &mut Criterion) {
|
||||
let files = match setup_once() {
|
||||
Ok(files) => files,
|
||||
let (sp, _sf) = match setup_once() {
|
||||
Ok(result) => result,
|
||||
Err(e) => {
|
||||
eprintln!("⚠ Skipping ordering benchmarks: {}", e);
|
||||
eprintln!("Skipping ordering benchmarks: {}", e);
|
||||
return;
|
||||
}
|
||||
};
|
||||
|
||||
let guard = sp.read().unwrap();
|
||||
let picker = guard.as_ref().unwrap();
|
||||
|
||||
let mut group = c.benchmark_group("ordering");
|
||||
group.sample_size(100);
|
||||
|
||||
let query = "controller";
|
||||
let parser = QueryParser::default();
|
||||
let parsed_controller = parser.parse("controller");
|
||||
let parsed_mod = parser.parse("mod");
|
||||
|
||||
// Benchmark normal order (descending)
|
||||
group.bench_function("normal_order", |b| {
|
||||
b.iter(|| {
|
||||
let results = FilePicker::fuzzy_search(
|
||||
black_box(&files),
|
||||
black_box(query),
|
||||
let results = picker.fuzzy_search(
|
||||
black_box(&parsed_controller),
|
||||
None,
|
||||
FuzzySearchOptions {
|
||||
max_threads: 4,
|
||||
current_file: None,
|
||||
|
||||
project_path: None,
|
||||
last_same_query_match: None,
|
||||
|
||||
combo_boost_score_multiplier: 100,
|
||||
min_combo_count: 3,
|
||||
pagination: PaginationArgs {
|
||||
@@ -443,15 +402,14 @@ fn bench_search_ordering(c: &mut Criterion) {
|
||||
// Benchmark reverse order (ascending)
|
||||
group.bench_function("reverse_order", |b| {
|
||||
b.iter(|| {
|
||||
let results = FilePicker::fuzzy_search(
|
||||
black_box(&files),
|
||||
black_box(query),
|
||||
let results = picker.fuzzy_search(
|
||||
black_box(&parsed_controller),
|
||||
None,
|
||||
FuzzySearchOptions {
|
||||
max_threads: 4,
|
||||
current_file: None,
|
||||
|
||||
project_path: None,
|
||||
last_same_query_match: None,
|
||||
|
||||
combo_boost_score_multiplier: 100,
|
||||
min_combo_count: 3,
|
||||
pagination: PaginationArgs {
|
||||
@@ -467,15 +425,14 @@ fn bench_search_ordering(c: &mut Criterion) {
|
||||
// Benchmark with large result set
|
||||
group.bench_function("normal_order_large", |b| {
|
||||
b.iter(|| {
|
||||
let results = FilePicker::fuzzy_search(
|
||||
black_box(&files),
|
||||
black_box("mod"),
|
||||
let results = picker.fuzzy_search(
|
||||
black_box(&parsed_mod),
|
||||
None,
|
||||
FuzzySearchOptions {
|
||||
max_threads: 4,
|
||||
current_file: None,
|
||||
|
||||
project_path: None,
|
||||
last_same_query_match: None,
|
||||
|
||||
combo_boost_score_multiplier: 100,
|
||||
min_combo_count: 3,
|
||||
pagination: PaginationArgs {
|
||||
@@ -490,15 +447,14 @@ fn bench_search_ordering(c: &mut Criterion) {
|
||||
|
||||
group.bench_function("reverse_order_large", |b| {
|
||||
b.iter(|| {
|
||||
let results = FilePicker::fuzzy_search(
|
||||
black_box(&files),
|
||||
black_box("mod"),
|
||||
let results = picker.fuzzy_search(
|
||||
black_box(&parsed_mod),
|
||||
None,
|
||||
FuzzySearchOptions {
|
||||
max_threads: 4,
|
||||
current_file: None,
|
||||
|
||||
project_path: None,
|
||||
last_same_query_match: None,
|
||||
|
||||
combo_boost_score_multiplier: 100,
|
||||
min_combo_count: 3,
|
||||
pagination: PaginationArgs {
|
||||
@@ -514,15 +470,14 @@ fn bench_search_ordering(c: &mut Criterion) {
|
||||
// Benchmark with small result set
|
||||
group.bench_function("normal_order_small", |b| {
|
||||
b.iter(|| {
|
||||
let results = FilePicker::fuzzy_search(
|
||||
black_box(&files),
|
||||
black_box("controller"),
|
||||
let results = picker.fuzzy_search(
|
||||
black_box(&parsed_controller),
|
||||
None,
|
||||
FuzzySearchOptions {
|
||||
max_threads: 4,
|
||||
current_file: None,
|
||||
|
||||
project_path: None,
|
||||
last_same_query_match: None,
|
||||
|
||||
combo_boost_score_multiplier: 100,
|
||||
min_combo_count: 3,
|
||||
pagination: PaginationArgs {
|
||||
@@ -537,15 +492,14 @@ fn bench_search_ordering(c: &mut Criterion) {
|
||||
|
||||
group.bench_function("reverse_order_small", |b| {
|
||||
b.iter(|| {
|
||||
let results = FilePicker::fuzzy_search(
|
||||
black_box(&files),
|
||||
black_box("controller"),
|
||||
let results = picker.fuzzy_search(
|
||||
black_box(&parsed_controller),
|
||||
None,
|
||||
FuzzySearchOptions {
|
||||
max_threads: 4,
|
||||
current_file: None,
|
||||
|
||||
project_path: None,
|
||||
last_same_query_match: None,
|
||||
|
||||
combo_boost_score_multiplier: 100,
|
||||
min_combo_count: 3,
|
||||
pagination: PaginationArgs {
|
||||
@@ -563,32 +517,36 @@ fn bench_search_ordering(c: &mut Criterion) {
|
||||
|
||||
/// Benchmark pagination: first page vs deep page
|
||||
fn bench_pagination_performance(c: &mut Criterion) {
|
||||
let files = match setup_once() {
|
||||
Ok(files) => files,
|
||||
let (sp, _sf) = match setup_once() {
|
||||
Ok(result) => result,
|
||||
Err(e) => {
|
||||
eprintln!("⚠ Skipping pagination benchmarks: {}", e);
|
||||
eprintln!("Skipping pagination benchmarks: {}", e);
|
||||
return;
|
||||
}
|
||||
};
|
||||
|
||||
let guard = sp.read().unwrap();
|
||||
let picker = guard.as_ref().unwrap();
|
||||
|
||||
let mut group = c.benchmark_group("pagination");
|
||||
group.sample_size(100);
|
||||
|
||||
let query = "mod";
|
||||
let parser = QueryParser::default();
|
||||
let parsed = parser.parse(query);
|
||||
let page_size = 40;
|
||||
|
||||
// Benchmark first page (uses partial sort optimization)
|
||||
group.bench_function("page_0_size_40", |b| {
|
||||
b.iter(|| {
|
||||
let results = FilePicker::fuzzy_search(
|
||||
black_box(&files),
|
||||
black_box(query),
|
||||
let results = picker.fuzzy_search(
|
||||
black_box(&parsed),
|
||||
None,
|
||||
FuzzySearchOptions {
|
||||
max_threads: 4,
|
||||
current_file: None,
|
||||
|
||||
project_path: None,
|
||||
last_same_query_match: None,
|
||||
|
||||
combo_boost_score_multiplier: 100,
|
||||
min_combo_count: 3,
|
||||
pagination: PaginationArgs {
|
||||
@@ -604,15 +562,14 @@ fn bench_pagination_performance(c: &mut Criterion) {
|
||||
// Benchmark 10th page (requires full sort, no optimization)
|
||||
group.bench_function("page_10_size_40", |b| {
|
||||
b.iter(|| {
|
||||
let results = FilePicker::fuzzy_search(
|
||||
black_box(&files),
|
||||
black_box(query),
|
||||
let results = picker.fuzzy_search(
|
||||
black_box(&parsed),
|
||||
None,
|
||||
FuzzySearchOptions {
|
||||
max_threads: 4,
|
||||
current_file: None,
|
||||
|
||||
project_path: None,
|
||||
last_same_query_match: None,
|
||||
|
||||
combo_boost_score_multiplier: 100,
|
||||
min_combo_count: 3,
|
||||
pagination: PaginationArgs {
|
||||
@@ -628,15 +585,14 @@ fn bench_pagination_performance(c: &mut Criterion) {
|
||||
// Benchmark 50th page (even deeper pagination)
|
||||
group.bench_function("page_50_size_40", |b| {
|
||||
b.iter(|| {
|
||||
let results = FilePicker::fuzzy_search(
|
||||
black_box(&files),
|
||||
black_box(query),
|
||||
let results = picker.fuzzy_search(
|
||||
black_box(&parsed),
|
||||
None,
|
||||
FuzzySearchOptions {
|
||||
max_threads: 4,
|
||||
current_file: None,
|
||||
|
||||
project_path: None,
|
||||
last_same_query_match: None,
|
||||
|
||||
combo_boost_score_multiplier: 100,
|
||||
min_combo_count: 3,
|
||||
pagination: PaginationArgs {
|
||||
@@ -652,15 +608,68 @@ fn bench_pagination_performance(c: &mut Criterion) {
|
||||
group.finish();
|
||||
}
|
||||
|
||||
/// Benchmark grep search via the FilePicker public API
|
||||
fn bench_grep_search(c: &mut Criterion) {
|
||||
let (sp, _sf) = match setup_once() {
|
||||
Ok(result) => result,
|
||||
Err(e) => {
|
||||
eprintln!("Skipping grep benchmarks: {}", e);
|
||||
return;
|
||||
}
|
||||
};
|
||||
|
||||
let guard = sp.read().unwrap();
|
||||
let picker = guard.as_ref().unwrap();
|
||||
|
||||
let mut group = c.benchmark_group("grep");
|
||||
group.sample_size(50);
|
||||
|
||||
let options = GrepSearchOptions {
|
||||
max_file_size: 10 * 1024 * 1024,
|
||||
max_matches_per_file: 0,
|
||||
smart_case: true,
|
||||
file_offset: 0,
|
||||
page_limit: 100,
|
||||
mode: GrepMode::PlainText,
|
||||
time_budget_ms: 0,
|
||||
before_context: 0,
|
||||
after_context: 0,
|
||||
classify_definitions: false,
|
||||
trim_whitespace: false,
|
||||
abort_signal: None,
|
||||
};
|
||||
|
||||
let test_queries = vec![
|
||||
("common", "struct"),
|
||||
("specific", "DEFINE_MUTEX"),
|
||||
("path_filter", "*.h mutex"),
|
||||
];
|
||||
|
||||
let grep_parser = fff::QueryParser::new(fff::GrepConfig);
|
||||
|
||||
for (name, query) in &test_queries {
|
||||
let parsed = grep_parser.parse(query);
|
||||
|
||||
group.bench_with_input(BenchmarkId::new("grep", name), query, |b, _| {
|
||||
b.iter(|| {
|
||||
let result = picker.grep(black_box(&parsed), black_box(&options));
|
||||
result.matches.len()
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
group.finish();
|
||||
}
|
||||
|
||||
criterion_group!(
|
||||
benches,
|
||||
bench_indexing,
|
||||
bench_search_queries,
|
||||
bench_search_thread_scaling,
|
||||
bench_search_result_limits,
|
||||
bench_search_scalability,
|
||||
bench_search_ordering,
|
||||
bench_pagination_performance,
|
||||
bench_grep_search,
|
||||
);
|
||||
|
||||
criterion_main!(benches);
|
||||
@@ -0,0 +1,242 @@
|
||||
use criterion::{BenchmarkId, Criterion, black_box, criterion_group, criterion_main};
|
||||
use fff::file_picker::{FFFMode, FilePicker};
|
||||
use fff::{
|
||||
FilePickerOptions, GrepMode, GrepSearchOptions, SharedFilePicker, SharedFrecency,
|
||||
parse_grep_query,
|
||||
};
|
||||
use std::sync::OnceLock;
|
||||
use std::time::Duration;
|
||||
|
||||
struct TestData {
|
||||
shared_picker: SharedFilePicker,
|
||||
}
|
||||
|
||||
static SETUP: OnceLock<TestData> = OnceLock::new();
|
||||
|
||||
fn big_repo_path() -> String {
|
||||
if let Some(path) = std::env::var_os("BIG_REPO_PATH") {
|
||||
return path.to_string_lossy().into_owned();
|
||||
}
|
||||
|
||||
let candidates = ["./big-repo", "../../big-repo"];
|
||||
for p in &candidates {
|
||||
if std::path::Path::new(p).exists() {
|
||||
return p.to_string();
|
||||
}
|
||||
}
|
||||
panic!(
|
||||
"./big-repo not found. Run from workspace root:\n \
|
||||
git clone --depth 1 https://github.com/torvalds/linux.git big-repo"
|
||||
);
|
||||
}
|
||||
|
||||
fn setup() -> &'static TestData {
|
||||
SETUP.get_or_init(|| {
|
||||
let path = big_repo_path();
|
||||
let shared_picker = SharedFilePicker::default();
|
||||
let shared_frecency = SharedFrecency::default();
|
||||
|
||||
eprintln!("Initializing FilePicker for {:?}...", path);
|
||||
FilePicker::new_with_shared_state(
|
||||
shared_picker.clone(),
|
||||
shared_frecency.clone(),
|
||||
FilePickerOptions {
|
||||
base_path: path,
|
||||
enable_mmap_cache: true,
|
||||
enable_content_indexing: true,
|
||||
mode: FFFMode::Neovim,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.expect("create picker");
|
||||
|
||||
eprintln!("Waiting for scan completion...");
|
||||
shared_picker.wait_for_scan(Duration::from_secs(120));
|
||||
|
||||
eprintln!("Waiting for warmup (bigram index)...");
|
||||
loop {
|
||||
let guard = shared_picker.read().expect("read lock");
|
||||
let picker = guard.as_ref().expect("picker present");
|
||||
let progress = picker.get_scan_progress();
|
||||
if progress.is_warmup_complete {
|
||||
let file_count = picker.get_files().len();
|
||||
eprintln!("Ready: {} files indexed, bigram built", file_count);
|
||||
break;
|
||||
}
|
||||
drop(guard);
|
||||
std::thread::sleep(Duration::from_millis(100));
|
||||
}
|
||||
|
||||
TestData { shared_picker }
|
||||
})
|
||||
}
|
||||
|
||||
fn setup_cold() -> SharedFilePicker {
|
||||
let path = big_repo_path();
|
||||
let shared_picker = SharedFilePicker::default();
|
||||
let shared_frecency = SharedFrecency::default();
|
||||
|
||||
FilePicker::new_with_shared_state(
|
||||
shared_picker.clone(),
|
||||
shared_frecency.clone(),
|
||||
FilePickerOptions {
|
||||
base_path: path,
|
||||
enable_mmap_cache: false,
|
||||
enable_content_indexing: false,
|
||||
mode: FFFMode::Neovim,
|
||||
watch: false,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.expect("create picker");
|
||||
|
||||
shared_picker.wait_for_scan(Duration::from_secs(120));
|
||||
shared_picker
|
||||
}
|
||||
|
||||
fn plain_options() -> GrepSearchOptions {
|
||||
GrepSearchOptions {
|
||||
max_file_size: 10 * 1024 * 1024,
|
||||
max_matches_per_file: 200,
|
||||
smart_case: true,
|
||||
file_offset: 0,
|
||||
page_limit: 50,
|
||||
mode: GrepMode::PlainText,
|
||||
time_budget_ms: 0,
|
||||
before_context: 0,
|
||||
after_context: 0,
|
||||
classify_definitions: false,
|
||||
trim_whitespace: false,
|
||||
..Default::default()
|
||||
}
|
||||
}
|
||||
|
||||
fn fuzzy_options() -> GrepSearchOptions {
|
||||
GrepSearchOptions {
|
||||
mode: GrepMode::Fuzzy,
|
||||
..plain_options()
|
||||
}
|
||||
}
|
||||
|
||||
const PLAIN_QUERIES: &[(&str, &str)] = &[
|
||||
("2char_if", "if"),
|
||||
("common_return", "return"),
|
||||
("func_mutex_lock", "mutex_lock"),
|
||||
("struct_inode_ops", "inode_operations"),
|
||||
("define_MODULE_LICENSE", "MODULE_LICENSE"),
|
||||
("rare_phylink_ethtool", "phylink_ethtool"),
|
||||
("include", "#include"),
|
||||
("comment_TODO", "TODO"),
|
||||
("type_struct_file", "struct file"),
|
||||
("error_EINVAL", "err = -EINVAL"),
|
||||
("long_static_int_init", "static int __init"),
|
||||
("very_common_int", "int"),
|
||||
("single_char_x", "x"),
|
||||
("path_printk_c", "printk *.c"),
|
||||
("dir_mutex_kernel", "mutex /kernel/"),
|
||||
];
|
||||
|
||||
const FUZZY_QUERIES: &[(&str, &str)] = &[
|
||||
("exact_mutex_lock", "mutex_lock"),
|
||||
("typo_mutx_lock", "mutx_lock"),
|
||||
("camel_InodeOps", "InodeOps"),
|
||||
("abbrev_sched_rt", "sched_rt"),
|
||||
("short_kfr", "kfr"),
|
||||
("common_return", "return"),
|
||||
("define_MODULE_LICENSE", "MODULE_LICENSE"),
|
||||
("struct_file_ops", "file_operations"),
|
||||
("long_static_int_init", "static_int_init"),
|
||||
("path_printk_c", "printk *.c"),
|
||||
];
|
||||
|
||||
fn bench_plain_warm(c: &mut Criterion) {
|
||||
let data = setup();
|
||||
let opts = plain_options();
|
||||
|
||||
let mut group = c.benchmark_group("plain_warm");
|
||||
group.sample_size(30);
|
||||
group.warm_up_time(Duration::from_secs(2));
|
||||
group.measurement_time(Duration::from_secs(5));
|
||||
|
||||
for (name, query) in PLAIN_QUERIES {
|
||||
group.bench_with_input(BenchmarkId::from_parameter(name), query, |b, q| {
|
||||
let guard = data.shared_picker.read().expect("read lock");
|
||||
let picker = guard.as_ref().expect("picker present");
|
||||
b.iter(|| {
|
||||
let parsed = parse_grep_query(q);
|
||||
black_box(picker.grep(&parsed, &opts))
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
group.finish();
|
||||
}
|
||||
|
||||
fn bench_fuzzy_warm(c: &mut Criterion) {
|
||||
let data = setup();
|
||||
let opts = fuzzy_options();
|
||||
|
||||
let mut group = c.benchmark_group("fuzzy_warm");
|
||||
group.sample_size(10);
|
||||
group.warm_up_time(Duration::from_secs(2));
|
||||
group.measurement_time(Duration::from_secs(8));
|
||||
|
||||
for (name, query) in FUZZY_QUERIES {
|
||||
group.bench_with_input(BenchmarkId::from_parameter(name), query, |b, q| {
|
||||
let guard = data.shared_picker.read().expect("read lock");
|
||||
let picker = guard.as_ref().expect("picker present");
|
||||
b.iter(|| {
|
||||
let parsed = parse_grep_query(q);
|
||||
black_box(picker.grep(&parsed, &opts))
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
group.finish();
|
||||
}
|
||||
|
||||
fn bench_plain_cold(c: &mut Criterion) {
|
||||
let _ = setup();
|
||||
let opts = plain_options();
|
||||
|
||||
let queries: &[(&str, &str)] = &[
|
||||
("2char_if", "if"),
|
||||
("common_return", "return"),
|
||||
("func_mutex_lock", "mutex_lock"),
|
||||
("struct_inode_ops", "inode_operations"),
|
||||
("define_MODULE_LICENSE", "MODULE_LICENSE"),
|
||||
("rare_phylink_ethtool", "phylink_ethtool"),
|
||||
("long_static_int_init", "static int __init"),
|
||||
];
|
||||
|
||||
let mut group = c.benchmark_group("plain_cold");
|
||||
group.sample_size(10);
|
||||
group.warm_up_time(Duration::from_millis(500));
|
||||
group.measurement_time(Duration::from_secs(10));
|
||||
|
||||
for (name, query) in queries {
|
||||
group.bench_with_input(BenchmarkId::from_parameter(name), query, |b, q| {
|
||||
b.iter_with_setup(
|
||||
|| setup_cold(),
|
||||
|cold_picker| {
|
||||
let guard = cold_picker.read().expect("read lock");
|
||||
let picker = guard.as_ref().expect("picker present");
|
||||
let parsed = parse_grep_query(q);
|
||||
let result = picker.grep(&parsed, &opts);
|
||||
black_box(result.matches.len())
|
||||
},
|
||||
);
|
||||
});
|
||||
}
|
||||
|
||||
group.finish();
|
||||
}
|
||||
|
||||
criterion_group!(
|
||||
benches,
|
||||
bench_plain_warm,
|
||||
bench_fuzzy_warm,
|
||||
bench_plain_cold,
|
||||
);
|
||||
|
||||
criterion_main!(benches);
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user