Compare commits

..
Author SHA1 Message Date
Gergo MagyarandClaude Opus 4.8 e766cedd1a fix(eval): drop tags from the benchmark's per-arm clone
Every benchmark-arm session failed with "sanitized graph snapshot
preparation failed: clone has more than 1024 references; refusing
incomplete sanitization" (confirmed via a real workflow_dispatch run,
29738099937, after the prior activation fixes let the proposer succeed
end-to-end for the first time).

make_worktree() creates each arm's throwaway clone with a plain `git
clone`, which inherits every tag and branch from the source. This repo's
history has grown to 1144 tags (a v1.6.9-rc.N release-candidate series)
out of 1650 total refs, exceeding oracle_assets.MAX_CLONE_REFS=1024 -- a
fail-closed guard in sanitize_clone_for_hidden_oracles() that refuses to
proceed unless it can enumerate and delete every ref before handing a
sanitized snapshot to a benchmark session (so an agent can never discover
oracle answers via a ref the sanitization missed).

`ref` at every call site (evolve.py, runner.py, sanitized_graph.py) is
always a bare SHA or the literal "HEAD", never a branch name, so
`--single-branch --branch <ref>` isn't viable (git clone's --branch
requires a name). Tags are never used by the checkout fallback or by
sanitization's own delete-everything behavior, so dropping them via
--no-tags removes the 1144-ref majority without touching branch-fetch
behavior or the existing ref/origin-ref checkout fallback, and without
weakening MAX_CLONE_REFS itself.

Verified against the real repository (not just the test fixture): cloning
/workspace (1650 refs, 1144 tags) via the fixed make_worktree() now
produces a clone with 237 total refs and 0 tags.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01Va5uu9Ar3e45QZ5xFsG4AZ
2026-07-20 11:36:58 +00:00
157 changed files with 795 additions and 7906 deletions
+1 -1
View File
@@ -6,7 +6,7 @@
"plugins": [
{
"name": "gitnexus",
"version": "1.6.10-rc.94",
"version": "1.6.9",
"source": {
"source": "local",
"path": "./gitnexus-claude-plugin"
+1 -1
View File
@@ -11,7 +11,7 @@
"plugins": [
{
"name": "gitnexus",
"version": "1.6.10-rc.94",
"version": "1.6.9",
"source": "./gitnexus-claude-plugin",
"description": "Code intelligence powered by a knowledge graph. Provides execution flow tracing, blast radius analysis, and augmented search across your codebase."
}
-5
View File
@@ -1,5 +0,0 @@
# Custom self-hosted runner labels actionlint can't discover on its own.
# gitnexus-evolution: the skill-evolution EC2 runner (infra/gitnexus-evolution/).
self-hosted-runner:
labels:
- gitnexus-evolution
+1 -1
View File
@@ -11,7 +11,7 @@
"@anthropic-ai/claude-code": "2.1.214"
},
"engines": {
"node": "22.18.0"
"node": "22.16.0"
}
},
"node_modules/@anthropic-ai/claude-code": {
+1 -1
View File
@@ -3,7 +3,7 @@
"version": "0.0.0",
"private": true,
"engines": {
"node": "22.18.0"
"node": "22.16.0"
},
"dependencies": {
"@anthropic-ai/claude-code": "2.1.214"
+1 -1
View File
@@ -11,7 +11,7 @@
"gitnexus": "1.6.9"
},
"engines": {
"node": "22.18.0"
"node": "22.16.0"
}
},
"node_modules/@emnapi/runtime": {
+1 -1
View File
@@ -3,7 +3,7 @@
"private": true,
"version": "1.0.0",
"engines": {
"node": "22.18.0"
"node": "22.16.0"
},
"dependencies": {
"gitnexus": "1.6.9"
@@ -352,7 +352,7 @@ jobs:
with:
persist-credentials: false # this job uploads artifacts (artipacked)
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
- uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0
with:
node-version: 22
+2 -2
View File
@@ -39,7 +39,7 @@ jobs:
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
with:
persist-credentials: false
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
- uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0
with:
node-version: 22
- name: Unit-test the host->container config transforms
@@ -60,7 +60,7 @@ jobs:
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
with:
persist-credentials: false
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
- uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0
with:
node-version: 22
# Builds the image the same way a developer's "Reopen in Container" does.
+2 -2
View File
@@ -14,7 +14,7 @@ jobs:
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
with:
persist-credentials: false
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
- uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0
with:
node-version: 22
cache: npm
@@ -29,7 +29,7 @@ jobs:
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
with:
persist-credentials: false
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
- uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0
with:
node-version: 22
cache: npm
+24 -31
View File
@@ -46,7 +46,7 @@ jobs:
with:
path: ~/.lbdb/extension
key: lbug-fts-${{ runner.os }}-${{ hashFiles('gitnexus/package-lock.json') }}
- name: Ensure FTS + VECTOR extensions installed
- name: Ensure FTS extension installed
run: npx tsx scripts/ensure-fts.ts
working-directory: gitnexus
- name: Run sharded tests with coverage (blob)
@@ -205,10 +205,6 @@ jobs:
# tsx-on-source path in CI (both entry points stay covered).
env:
GITNEXUS_REQUIRE_FTS: '1'
# #2623: the win32 VECTOR gate is gone, so the vector suites genuinely
# run here — require the extension so an unavailable VECTOR is a loud
# failure, never a silent skip (same contract as GITNEXUS_REQUIRE_FTS).
GITNEXUS_REQUIRE_VECTOR: '1'
GITNEXUS_E2E_CLI: dist
# #2449: hosted Windows runners intermittently push the busiest shard past
# the default 15-minute watchdog. 20 minutes restores real headroom while
@@ -223,21 +219,19 @@ jobs:
- uses: ./.github/actions/setup-gitnexus
with:
build: 'true'
# Warm-cache the installed LadybugDB FTS + VECTOR extensions
# (~/.lbdb/extension) per OS + lockfile so a warm run skips the network
# install entirely, and the parallel shards share one download across
# runs. Pure reliability/speed: on a cache miss the tests self-install on
# demand (see test/helpers/fts-availability.ts), so a miss just falls
# back to install — never a correctness dependency. Keyed by lockfile
# hash so a LadybugDB version bump re-installs; per-OS because the
# extensions are native binaries. (Key name kept as lbug-fts for cache
# continuity — the path covers every extension in the shared home.)
# Warm-cache the installed LadybugDB FTS extension (~/.lbdb/extension) per
# OS + lockfile so a warm run skips the network install entirely, and the
# parallel shards share one download across runs. Pure reliability/speed:
# on a cache miss the tests self-install FTS on demand (see
# test/helpers/fts-availability.ts), so a miss just falls back to install —
# never a correctness dependency. Keyed by lockfile hash so a LadybugDB
# version bump re-installs; per-OS because the extension is a native binary.
- name: Cache LadybugDB FTS extension
uses: actions/cache@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v5
with:
path: ~/.lbdb/extension
key: lbug-fts-${{ runner.os }}-${{ hashFiles('gitnexus/package-lock.json') }}
- name: Ensure FTS + VECTOR extensions installed
- name: Ensure FTS extension installed
run: npx tsx scripts/ensure-fts.ts
working-directory: gitnexus
- name: Run platform-sensitive tests
@@ -384,16 +378,15 @@ jobs:
"$PREFIX/bin/gitnexus" --version
fi
# Node engines-floor gate (#2372). A module that statically names an API
# newer than the supported floor (e.g. `module.registerHooks`, added in
# 22.15) fails to LINK on the floor — a class vitest/tsx transforms
# structurally mask, and the default `node-version: 22` (resolves to latest)
# never hits. Build the dist on 22.x, then import-link every module R1 names
# as a load surface on the pinned engines floor (22.18.0, per package.json
# `engines: ^22.18.0 || >=24.11.0`) so a regression fails here instead of
# shipping to users on the minimum supported Node.
# Node engines-floor gate (#2372). The embedding resolvers statically named
# `module.registerHooks`, which only exists on Node >= 22.15 / >= 23.5, so on
# the supported floor (engines: >=22.0.0) those ESM modules failed to LINK —
# a class vitest/tsx transforms structurally mask, and the default
# `node-version: 22` (resolves to latest) never hits. Build the dist on 22.x,
# then import-link every module R1 names as a load surface on a pinned 22.14
# so a regression fails here instead of shipping to users on that Node range.
node-floor-compat:
name: node floor compat (22.18)
name: node floor compat (22.14)
runs-on: ubuntu-latest
timeout-minutes: 15
steps:
@@ -402,7 +395,7 @@ jobs:
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
with:
persist-credentials: false
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
- uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0
with:
node-version: '22'
cache: npm
@@ -420,16 +413,16 @@ jobs:
# Switch to the engines-floor Node AFTER building — native deps built on
# 22.x load across the whole 22.x ABI line, and nothing installs after this
# (so no package-manager cache is needed).
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
- uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0
with:
node-version: '22.18.0'
node-version: '22.14.0'
package-manager-cache: false
- name: Import-link the built dist on Node 22.18
- name: Import-link the built dist on Node 22.14
shell: bash
run: |
set -euo pipefail
node --version
node --version | grep -q '^v22\.18\.' || { echo "expected Node 22.18.x" >&2; exit 1; }
node --version | grep -q '^v22\.14\.' || { echo "expected Node 22.14.x" >&2; exit 1; }
for m in \
core/embeddings/runtime-install \
core/embeddings/onnxruntime-node-resolver \
@@ -561,9 +554,9 @@ jobs:
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
with:
persist-credentials: false
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
- uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0
with:
node-version: '22.18.0'
node-version: '22.16.0'
cache: npm
cache-dependency-path: |
gitnexus/package-lock.json
+6 -6
View File
@@ -323,9 +323,9 @@ jobs:
- name: Set up pinned Node.js
id: setup-node
if: steps.context.outputs.ready == 'true'
uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0
with:
node-version: '22.18.0'
node-version: '22.16.0'
- name: Install and preflight Claude subprocess isolation
id: isolation
@@ -377,7 +377,7 @@ jobs:
.github/claude-canary-runtime/package-lock.json \
"${runtime_dir}/package-lock.json"
printf '%s\n' 'registry=https://registry.npmjs.org/' 'audit=false' 'fund=false' > "${npmrc}"
test "$(node --version)" = 'v22.18.0'
test "$(node --version)" = 'v22.16.0'
test "$(uname -m)" = 'x86_64'
# The trusted lock and these independent receipts pin both the thin
@@ -398,7 +398,7 @@ jobs:
if (
lock.lockfileVersion !== 3 ||
lock.packages?.['']?.dependencies?.['@anthropic-ai/claude-code'] !== '2.1.214' ||
lock.packages?.['']?.engines?.node !== '22.18.0'
lock.packages?.['']?.engines?.node !== '22.16.0'
) {
throw new Error('Claude runtime lock root is not exact');
}
@@ -506,7 +506,7 @@ jobs:
install -m 0600 .github/gitnexus-review-runtime/package.json "${runtime_dir}/package.json"
install -m 0600 .github/gitnexus-review-runtime/package-lock.json "${runtime_dir}/package-lock.json"
printf '%s\n' 'registry=https://registry.npmjs.org/' 'audit=false' 'fund=false' > "${npmrc}"
test "$(node --version)" = 'v22.18.0'
test "$(node --version)" = 'v22.16.0'
npm ci \
--prefix "${runtime_dir}" \
--userconfig "${npmrc}" \
@@ -1241,7 +1241,7 @@ jobs:
CLAUDE_CONFIG_DIR: ${{ runner.temp }}/gitnexus-review-claude-config
CLAUDE_WORKING_DIR: ${{ runner.temp }}/gitnexus-review-control
NPM_CONFIG_IGNORE_SCRIPTS: 'true'
NODE_VERSION: '22.18.0'
NODE_VERSION: '22.16.0'
with:
claude_code_oauth_token: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN }}
path_to_claude_code_executable: ${{ runner.temp }}/gitnexus-review-claude-runtime/node_modules/@anthropic-ai/claude-code/bin/claude.exe
+5 -27
View File
@@ -12,34 +12,12 @@
# App that opens the promotion PR). The Mint-App-Token step hard-fails
# without them once a promotion is detected. Verify the App installation
# is scoped to this repo with only Contents: RW + Pull requests: RW.
# [x] Create the protected Environment `gitnexus-evolution` with a
# [ ] Create the protected Environment `gitnexus-evolution` with a
# deployment-branch rule restricting it to `main`, and ideally scope the
# three secrets above to that Environment. workflow_dispatch runs this
# workflow (and eval/workflow_bench/evolve.py) from the *dispatched ref*,
# so this server-side rule — not a code-side guard the branch could edit
# away — is what stops a non-main branch from running with the secrets.
# [x] Register a self-hosted runner labeled `gitnexus-evolution` (a dedicated
# EC2 box works well). GitHub-hosted runners hard-cap job execution at 6
# hours, non-configurable — too short once a benchmark session actually
# invokes Skill/MCP tools for real. Self-hosted runners cap at 5 days
# instead. This job only ever runs on schedule/workflow_dispatch, never
# on fork-PR content, so the usual public-repo self-hosted-runner risk
# doesn't apply — still keep the box dedicated to this workflow, with
# outbound-only network access, and prefer on-demand over Spot (a Spot
# reclaim mid-run loses the same way a 6-hour timeout does). Instance,
# security group, and IAM setup are documented privately, not in this
# repo — publishing the exact topology of a real, live AWS account
# isn't safe to do in a public repo even without literal secrets.
# Accepted tradeoff: the box is stopped between runs (an EventBridge
# schedule starts it ~15min before the Saturday cron and stops it 24h
# later) but is not destroyed/recreated per run, so it isn't fully
# ephemeral — a compromise between the review-flagged ideal (re-image
# between runs, bounding how long the injected model API key could
# matter if the box were ever compromised some other way) and the added
# complexity of per-job ephemeral provisioning for a job that runs at
# most weekly. Revisit if run frequency increases or the threat model
# changes; stopping already bounds the exposure window to the job's own
# runtime on 1 day out of 7.
# [ ] Run workflow_dispatch once and confirm: containment preflight passes,
# the benchmark completes inside the job timeout, the results artifact
# uploads, and a promotion (if any) opens a well-formed PR.
@@ -99,13 +77,13 @@ jobs:
github.event_name == 'workflow_dispatch' ||
vars.GITNEXUS_EVOLUTION_ENABLED == 'true'
)
runs-on: [self-hosted, linux, x64, gitnexus-evolution]
runs-on: ubuntu-latest
# Gate promotion runs on a protected Environment. An admin must attach a
# deployment-branch rule (main only) and ideally scope the three secrets to
# it — server-side enforcement a dispatched non-main ref cannot bypass by
# editing its own workflow copy. See the activation checklist above.
environment: gitnexus-evolution
timeout-minutes: 1440 # self-hosted ceiling is 5 days (7200min); 24h is a generous margin over a single-generation serial run
timeout-minutes: 355 # ceiling just under GitHub's 360-minute hard cap
permissions:
contents: read # The promotion PR uses a short-lived App token minted below.
env:
@@ -130,9 +108,9 @@ jobs:
persist-credentials: false
fetch-depth: 0
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
- uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0
with:
node-version: '22.18.0'
node-version: '22.16.0'
cache: npm
cache-dependency-path: |
gitnexus/package-lock.json
+1 -1
View File
@@ -48,7 +48,7 @@ jobs:
with:
persist-credentials: false
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
- uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0
with:
node-version: 22
+1 -1
View File
@@ -59,7 +59,7 @@ jobs:
repository: ${{ github.event.pull_request.head.repo.full_name }}
persist-credentials: false
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
- uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0
with:
node-version: 22
cache: npm
+2 -2
View File
@@ -369,7 +369,7 @@ jobs:
exit 1
fi
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
- uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0
with:
# Node 24 ships with npm >= 11.5.x, which is the minimum that
# supports npm Trusted Publishing OIDC. Node 22 ships with npm
@@ -828,7 +828,7 @@ jobs:
fi
- name: Create GitHub Release
uses: softprops/action-gh-release@3d0d9888cb7fd7b750713d6e236d1fcb99157228 # v2
uses: softprops/action-gh-release@718ea10b132b3b2eba29c1007bb80653f286566b # v2
with:
tag_name: ${{ steps.vtag-gate.outputs.vtag }}
name: >-
+1 -1
View File
@@ -50,7 +50,7 @@ jobs:
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
with:
persist-credentials: false
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
- uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0
with:
node-version: '22'
cache: npm
+2 -1
View File
@@ -68,8 +68,9 @@ gitnexus-web/test-results/
eval/.coverage
eval/.hypothesis/
# Local docs — planning output (gitnexus-plan / gitnexus-work) stays local, not tracked
# Local docs (docs/plans/ stays tracked — gitnexus-plan output travels with the work)
docs/*
!docs/plans/
gitnexus/test/fixtures/mini-repo/*.md
gitnexus/test/fixtures/mini-repo/.claude
+1 -1
View File
@@ -13,7 +13,7 @@ This project uses the [PolyForm Noncommercial License 1.0.0](https://polyformpro
## Development setup
**Prerequisites:** Node.js — `gitnexus/` requires `^22.18.0 || >=24.11.0` and `gitnexus-web/` requires `^20.19.0 || >=22.12.0` (enforced via the `engines` field in each package). Use `nvm install` to match the local version.
**Prerequisites:** Node.js — `gitnexus/` requires `>=22.0.0` and `gitnexus-web/` requires `^20.19.0 || >=22.12.0` (enforced via the `engines` field in each package). Use `nvm install` to match the local version.
1. Clone the repository.
2. **Shared package:** `cd gitnexus-shared && npm install && npm run build`
+1 -2
View File
@@ -488,10 +488,9 @@ Most `analyze` knobs are also CLI flags (`--workers`, `--worker-timeout`, `--max
| `PROF_LBUG_LOAD` | unset | When `1`, emits one `[lbug-load prof]` summary line per `loadGraphToLbug` call breaking the graph-DB persistence wall into stages (`csv-emit` / `copy-nodes` / `copy-rels` / `fallback` / `total`) plus node & edge counts. Zero-cost when unset. | Attributing large-repo analyze wall time across CSV generation vs. LadybugDB `COPY` (issue #2203) — the analyze "emit" timing is the scope-resolution bucket, not this DB-write path. |
| `GITNEXUS_MAX_FILE_SIZE` | `512` (KB) | Walker skip threshold in KB. Hard cap is `32768` (tree-sitter buffer ceiling). Equivalent to `--max-file-size <kb>`. | Indexing repos with intentionally-large source files (generated parsers, vendored bundles) that should still be parsed. |
| `GITNEXUS_WORKER_SUB_BATCH_TIMEOUT_MS` | `30000` | Worker idle timeout in milliseconds before retry/fallback. Equivalent to `--worker-timeout <seconds>` × 1000. | Slow-parsing files (large minified JS, deeply-nested TS types) that legitimately need more than 30s. |
| `GITNEXUS_WORKER_READY_TIMEOUT_MS` | `5000` | Startup budget in milliseconds for a parse worker to load its grammar bindings and report `{type:'ready'}`. Slots that miss it are treated as startup crashes. | Slow or heavily loaded hosts where a full pool cold-starting concurrently needs more than 5s, and analyze aborts with "did not report ready within 5000ms". |
| `GITNEXUS_FTS_STEMMER` | `porter` | Stemmer used when rebuilding BM25/FTS indexes. Use `none` for CJK-heavy repositories, or a language stemmer such as `german`, `french`, or `spanish` for matching repository comments. Re-run `gitnexus analyze --repair-fts` after changing it. | Keyword search quality is poor for non-English comments or identifiers under English stemming. |
| `GITNEXUS_WAL_CHECKPOINT_THRESHOLD` | `67108864` (64 MiB) | LadybugDB WAL auto-checkpoint threshold in bytes. Equivalent to `--wal-checkpoint-threshold <bytes>`. `-1` keeps LadybugDB's stock threshold (~16 MiB). Larger thresholds reduce checkpoint frequency but increase the WAL size at rotation time — choose a smaller value on disk-constrained environments. | You need a larger or smaller WAL auto-checkpoint threshold for your analyze workload. |
| `GITNEXUS_LBUG_BUFFER_POOL_SIZE` | min(2 GiB, 80% RAM) | LadybugDB buffer-pool ceiling in bytes for every GitNexus database (analyze, MCP server, serve, group bridges). `0` restores LadybugDB's native unbounded default of 80% of system RAM; invalid values warn and fall back to the default (#2557). During `analyze` the pool is right-sized to the graph, scaled on non-4 KiB-page hosts by the page-size granule ratio up to min(2 GiB × pageSize/4 KiB, 80% RAM) (#2631); this env var overrides all of that as an absolute value. | A long-lived `gitnexus mcp` or a big incremental `analyze` uses too much memory, or a huge repo's working set genuinely needs a pool larger than 2 GiB. |
| `GITNEXUS_LBUG_BUFFER_POOL_SIZE` | min(2 GiB, 80% RAM) | LadybugDB buffer-pool ceiling in bytes for every GitNexus database (analyze, MCP server, serve, group bridges). `0` restores LadybugDB's native unbounded default of 80% of system RAM; invalid values warn and fall back to the default (#2557). | A long-lived `gitnexus mcp` or a big incremental `analyze` uses too much memory, or a huge repo's working set genuinely needs a pool larger than 2 GiB. |
| `GITNEXUS_LBUG_MAX_DB_SIZE` | `17179869184` (16 GiB) | Maximum size in bytes of a single LadybugDB database file — an mmap/disk-address-space ceiling, not a memory limit (it does not constrain the buffer pool). Invalid values silently fall back to the default. | Indexing a genuinely huge monorepo whose on-disk graph index approaches 16 GiB. |
| `GITNEXUS_WORKER_SUB_BATCH_MAX_BYTES` | `8388608` (8 MB) | Per-job byte budget the pool will send to a worker in one `postMessage`. | Very large individual files; mostly diagnostic — bumping past 8 MB risks structured-clone memory pressure. |
| `GITNEXUS_WORKER_MAX_RESPAWNS_PER_SLOT` | `3` | Max replacement spawns per worker slot before the slot is dropped from the active rotation. Bounds respawn loops on a chronically-crashing slot. | Hosts where a flaky worker should retry more (raise) or fail-fast (lower) before the slot is dropped. |
+2 -318
View File
@@ -4,7 +4,6 @@ from __future__ import annotations
import json
import os
import shutil
import stat
import subprocess
import sys
@@ -20,16 +19,10 @@ from workflow_bench.process_control import ManagedProcessResult, run_managed
from workflow_bench.proposer_sandbox import (
MAX_BUNDLE_BYTES,
MAX_EVIDENCE_FILE_BYTES,
SANDBOX_NODE,
SANDBOX_NODE_PREFIX,
VITE_TEMP_DIR,
SANDBOX_PATH,
SANDBOX_PYTHON3,
SANDBOX_SHELL_PREFIX,
SANDBOX_USER_SKILLS,
ReadOnlyMount,
SandboxError,
_runtime_mount_args,
build_claude_settings,
build_sandbox_environment,
prepare_sandbox,
@@ -168,28 +161,7 @@ def test_sandbox_command_has_minimal_mounts_and_no_host_root_bind(tmp_path: Path
check=False,
)
assert probe.returncode == 0, probe.stderr
assert probe.stdout == f"/home/agent|{SANDBOX_PATH}"
# The evidence-provenance.mjs plan-writer's PATH-scan trusts a Python 3
# candidate only if it (and its directory) is owned by root or by the
# current process — real /usr/bin/python3 is root-owned on the host,
# which surfaces as the kernel's overflow uid inside this
# --unshare-user sandbox (root itself is never mapped in). This wrapper
# is freshly created by the host process instead, so it's trusted, and
# it must still exec through to a real, working Python 3.
python3_index = argv.index(SANDBOX_PYTHON3)
assert argv[python3_index - 2] == "--ro-bind"
python3_wrapper = Path(argv[python3_index - 1])
assert stat.S_IMODE(python3_wrapper.stat().st_mode) == 0o500
version = subprocess.run(
[str(python3_wrapper), "-I", "-S", "-c", "import sys; print(sys.version_info[0])"],
text=True,
capture_output=True,
check=False,
)
assert version.returncode == 0, version.stderr
assert version.stdout.strip() == "3"
assert probe.stdout == "/home/agent|/opt/claude:/usr/local/bin:/usr/bin:/bin"
assert SANDBOX_USER_SKILLS in argv
user_skills_index = argv.index(SANDBOX_USER_SKILLS)
assert argv[user_skills_index - 2] == "--ro-bind"
@@ -197,223 +169,6 @@ def test_sandbox_command_has_minimal_mounts_and_no_host_root_bind(tmp_path: Path
assert not private_root.exists()
def test_runtime_mounts_bind_the_resolved_node_to_a_fresh_sandbox_path(monkeypatch) -> None:
# sanitized_graph.py and runner_sessions.py invoke the sandboxed graph CLI
# via SANDBOX_NODE. node's real host location varies (GitHub-hosted
# runner images happen to have one under /usr/local/bin; a self-hosted
# runner's actions/setup-node installs into its own tool-cache directory
# instead), so this must bind to a FRESH sandbox path like /opt/claude/...
# rather than anywhere under /usr, /bin, /lib, or /lib64: those are
# already read-only bound by this same function, and bwrap can't create
# a new mount-point file inside an already-read-only tree when the real
# path doesn't already exist there on the host (observed empirically:
# "bwrap: Can't create file at /usr/local/bin/node: Read-only file
# system" when this bind first targeted that path on a self-hosted
# runner where node isn't really there).
monkeypatch.setattr(
"workflow_bench.proposer_sandbox.shutil.which",
lambda name: "/opt/hostedtoolcache/node/22.18.0/x64/bin/node" if name == "node" else None,
)
args = _runtime_mount_args()
node_index = args.index("/opt/hostedtoolcache/node/22.18.0/x64/bin/node")
assert args[node_index - 1] == "--ro-bind"
assert args[node_index + 1] == SANDBOX_NODE
assert not any(SANDBOX_NODE.startswith(bound + "/") for bound in ("/usr", "/bin", "/lib", "/lib64"))
def test_runtime_mounts_bind_the_node_prefix_so_npx_and_npm_resolve(monkeypatch, tmp_path) -> None:
# npx and npm are not standalone binaries -- they are symlinks into
# ../lib/node_modules/npm/bin/*-cli.js -- so binding the sibling files is
# not enough; the install prefix carrying both bin/ and lib/node_modules
# has to be mounted. Without this, a self-hosted runner (where
# actions/setup-node installs into its own tool cache, outside /usr) gets
# a sandbox with node but no npx, and every task verify command dies with
# "/bin/sh: 1: npx: not found" -- all 18 runs of skill-evolution run
# 29861768554 did exactly that.
prefix = tmp_path / "hostedtoolcache" / "node" / "22.18.0" / "x64"
(prefix / "bin").mkdir(parents=True)
(prefix / "bin" / "node").write_text("#!/bin/sh\nexit 0\n")
(prefix / "lib" / "node_modules" / "npm" / "bin").mkdir(parents=True)
(prefix / "lib" / "node_modules" / "npm" / "bin" / "npx-cli.js").write_text("")
(prefix / "bin" / "npx").symlink_to("../lib/node_modules/npm/bin/npx-cli.js")
monkeypatch.setattr(
"workflow_bench.proposer_sandbox.shutil.which",
lambda name: str(prefix / "bin" / "node") if name == "node" else None,
)
args = _runtime_mount_args()
prefix_index = args.index(str(prefix))
assert args[prefix_index - 1] == "--ro-bind"
assert args[prefix_index + 1] == SANDBOX_NODE_PREFIX
# the single-binary bind stays: sanitized_graph.py and runner_sessions.py
# invoke SANDBOX_NODE directly.
node_index = args.index(str(prefix / "bin" / "node"))
assert args[node_index + 1] == SANDBOX_NODE
# and the prefix's bin/ must actually be on PATH for npx to resolve.
assert f"{SANDBOX_NODE_PREFIX}/bin" in SANDBOX_PATH.split(":")
def test_runtime_mounts_skip_the_prefix_bind_for_an_unrecognized_node_layout(monkeypatch, tmp_path) -> None:
# The prefix is derived from the node binary's path, so it must only be
# trusted when the layout really is <prefix>/bin/node carrying npm.
# Otherwise parent.parent names an unrelated ancestor: /opt/bin/node would
# bind ALL of /opt (every tool cache on a hosted runner) and a bare
# <dir>/node would bind <dir>'s parent -- an over-broad mount into a
# sandbox that runs untrusted model-authored code. The pre-existing
# real-Bubblewrap node canary builds exactly this bare <dir>/node shape.
bare = tmp_path / "toolcache"
bare.mkdir()
(bare / "node").write_text("#!/bin/sh\nexit 0\n")
monkeypatch.setattr(
"workflow_bench.proposer_sandbox.shutil.which",
lambda name: str(bare / "node") if name == "node" else None,
)
args = _runtime_mount_args()
assert SANDBOX_NODE_PREFIX not in args
assert str(tmp_path) not in args
# the node bind itself is unaffected -- SANDBOX_NODE still works.
assert args[args.index(str(bare / "node")) + 1] == SANDBOX_NODE
def test_runtime_mounts_skip_the_prefix_bind_without_npx_beside_node(monkeypatch, tmp_path) -> None:
# Right <prefix>/bin/node shape, but no working npx beside it: binding the
# prefix would widen the mount surface without making npx resolvable.
prefix = tmp_path / "x64"
(prefix / "bin").mkdir(parents=True)
(prefix / "bin" / "node").write_text("#!/bin/sh\nexit 0\n")
monkeypatch.setattr(
"workflow_bench.proposer_sandbox.shutil.which",
lambda name: str(prefix / "bin" / "node") if name == "node" else None,
)
args = _runtime_mount_args()
assert SANDBOX_NODE_PREFIX not in args
def test_runtime_mounts_bind_a_real_tool_cache_layout(monkeypatch, tmp_path) -> None:
# The positive counterpart: a genuine <prefix>/bin/node install carrying
# npm, outside the system trees, is bound so npx resolves.
prefix = tmp_path / "node" / "22.18.0" / "x64"
(prefix / "bin").mkdir(parents=True)
(prefix / "bin" / "node").write_text("#!/bin/sh\nexit 0\n")
(prefix / "lib" / "node_modules" / "npm" / "bin").mkdir(parents=True)
(prefix / "lib" / "node_modules" / "npm" / "bin" / "npx-cli.js").write_text("")
(prefix / "bin" / "npx").symlink_to("../lib/node_modules/npm/bin/npx-cli.js")
monkeypatch.setattr(
"workflow_bench.proposer_sandbox.shutil.which",
lambda name: str(prefix / "bin" / "node") if name == "node" else None,
)
args = _runtime_mount_args()
prefix_index = args.index(SANDBOX_NODE_PREFIX)
assert args[prefix_index - 2] == "--ro-bind"
assert args[prefix_index - 1] == str(prefix)
def test_runtime_mounts_skip_the_prefix_bind_when_it_is_already_bound(monkeypatch) -> None:
# On an image where node genuinely lives in /usr/local/bin, the prefix is
# /usr/local -- already inside the wholesale /usr read-only bind. Binding
# it again would be redundant and would needlessly widen the argv, so the
# containment surface stays minimal.
monkeypatch.setattr(
"workflow_bench.proposer_sandbox.shutil.which",
lambda name: "/usr/local/bin/node" if name == "node" else None,
)
args = _runtime_mount_args()
assert SANDBOX_NODE_PREFIX not in args
assert args[args.index("/usr/local/bin/node") + 1] == SANDBOX_NODE
def test_runtime_mounts_skip_the_node_bind_when_node_is_unresolvable(monkeypatch) -> None:
monkeypatch.setattr("workflow_bench.proposer_sandbox.shutil.which", lambda name: None)
args = _runtime_mount_args()
assert SANDBOX_NODE not in args
def test_node_modules_mounts_get_a_writable_vite_temp_overlay(tmp_path: Path) -> None:
# vite writes <node_modules>/.vite-temp/<config>.timestamp-*.mjs before
# loading a TypeScript config, so a read-only dependency mount makes vitest
# fail with EROFS before any test runs -- and every task verify command and
# every hidden oracle ends in "npx vitest run <test>". Reproduced on the
# self-hosted runner with npx bypassed entirely, proving it is independent
# of the node-prefix mount.
clone = tmp_path / "clone"
clone.mkdir()
deps = tmp_path / "deps"
deps.mkdir()
# task_assets.py captures this directory into the dependency snapshot; the
# overlay is gated on the mount source actually carrying it.
(deps / VITE_TEMP_DIR).mkdir()
executable = tmp_path / "executable"
executable.write_text("#!/bin/sh\nexit 0\n")
executable.chmod(0o755)
with prepare_sandbox(
clone=clone,
claude_bin=executable,
bwrap_bin=executable,
preflight=False,
read_only_mounts=(ReadOnlyMount(source=deps, target="/workspace/gitnexus/node_modules"),),
) as sandbox:
argv = sandbox.command_prefix
bind_index = argv.index("/workspace/gitnexus/node_modules")
assert argv[bind_index - 2 : bind_index + 1] == ["--ro-bind", str(deps), "/workspace/gitnexus/node_modules"]
overlay = f"/workspace/gitnexus/node_modules/{VITE_TEMP_DIR}"
overlay_index = argv.index(overlay)
assert argv[overlay_index - 1] == "--tmpfs"
# the overlay must come AFTER the read-only bind, or the bind would mask it
assert overlay_index > bind_index
def test_node_modules_mount_without_a_captured_vite_temp_gets_no_overlay(tmp_path: Path) -> None:
# The trusted GitNexus runtime mounts /opt/gitnexus/node_modules, whose
# source is the built runtime and does NOT carry a .vite-temp. bwrap cannot
# mkdir a mount point inside a read-only bind, so overlaying it would fail
# with "Can't mkdir .../node_modules/.vite-temp: Read-only file system".
# Regression for that CI failure: the overlay must fire only where the
# source actually contains the directory, not for every node_modules mount.
clone = tmp_path / "clone"
clone.mkdir()
runtime = tmp_path / "runtime-node-modules"
runtime.mkdir() # deliberately no .vite-temp
executable = tmp_path / "executable"
executable.write_text("#!/bin/sh\nexit 0\n")
executable.chmod(0o755)
with prepare_sandbox(
clone=clone,
claude_bin=executable,
bwrap_bin=executable,
preflight=False,
read_only_mounts=(ReadOnlyMount(source=runtime, target="/opt/gitnexus/node_modules"),),
) as sandbox:
argv = sandbox.command_prefix
assert "/opt/gitnexus/node_modules" in argv
assert not any(str(item).endswith(f"/{VITE_TEMP_DIR}") for item in argv)
def test_non_node_modules_mounts_get_no_vite_temp_overlay(tmp_path: Path) -> None:
# Scoped to dependency mounts: a hidden-oracle or skill mount stays wholly
# read-only, with no writable island inside it.
clone = tmp_path / "clone"
clone.mkdir()
other = tmp_path / "oracle"
other.mkdir()
executable = tmp_path / "executable"
executable.write_text("#!/bin/sh\nexit 0\n")
executable.chmod(0o755)
with prepare_sandbox(
clone=clone,
claude_bin=executable,
bwrap_bin=executable,
preflight=False,
read_only_mounts=(ReadOnlyMount(source=other, target="/workspace/.wfbench-oracle-abc"),),
) as sandbox:
argv = sandbox.command_prefix
assert not any(str(item).endswith(f"/{VITE_TEMP_DIR}") for item in argv)
def test_stricter_prefix_freezes_evaluated_skills_and_can_unshare_network(tmp_path: Path) -> None:
clone = tmp_path / "clone"
skill = clone / ".claude" / "skills" / "gitnexus-work"
@@ -442,78 +197,6 @@ def test_stricter_prefix_freezes_evaluated_skills_and_can_unshare_network(tmp_pa
assert prefix[user_index - 2] == "--ro-bind"
@pytest.mark.skipif(
os.environ.get("GITNEXUS_REQUIRE_BWRAP_CANARY") != "1",
reason="real Bubblewrap canary is mandatory in the named Ubuntu CI job",
)
def test_real_bubblewrap_runs_node_from_outside_the_bound_trees(tmp_path: Path, monkeypatch) -> None:
# Reproduces the self-hosted-runner failure directly: node resolved from
# a path outside /usr, /bin, /lib, /lib64 (actions/setup-node's own
# tool-cache convention) must still be reachable inside the sandbox at
# SANDBOX_NODE. A real node copied to a fresh, non-system location stands
# in for the tool-cache install; argv-construction tests alone can't
# catch a bwrap-level "Can't create file ...: Read-only file system"
# (the actual error this fix resolves), only a real bwrap invocation can.
real_node = shutil.which("node")
if not real_node:
pytest.skip("no node on PATH to relocate for this canary")
toolcache = tmp_path / "toolcache"
toolcache.mkdir()
relocated_node = toolcache / "node"
shutil.copy2(real_node, relocated_node)
relocated_node.chmod(0o755)
# Only fake "node"'s resolution -- prepare_sandbox's own bwrap/claude
# lookups (_resolve_executable) also go through shutil.which, and must
# keep resolving for real or preflight fails before the sandbox is even
# built.
real_which = shutil.which
monkeypatch.setattr(
"workflow_bench.proposer_sandbox.shutil.which",
lambda name: str(relocated_node) if name == "node" else real_which(name),
)
clone = tmp_path / "clone"
clone.mkdir()
with prepare_sandbox(clone=clone, claude_bin=Path(sys.executable), preflight=True) as sandbox:
result = sandbox.run([SANDBOX_NODE, "--version"], timeout=10)
assert result.ok, result.stderr_tail
@pytest.mark.skipif(
os.environ.get("GITNEXUS_REQUIRE_BWRAP_CANARY") != "1",
reason="real Bubblewrap canary is mandatory in the named Ubuntu CI job",
)
def test_real_bubblewrap_runs_npx_from_outside_the_bound_trees(tmp_path: Path, monkeypatch) -> None:
# The npx half of the self-hosted-runner failure. Relocating a real node
# INSTALL (bin/ + lib/node_modules, not just the binary) to a fresh path
# outside /usr, /bin, /lib and /lib64 reproduces actions/setup-node's
# tool-cache convention. Every task verify command is
# "cd gitnexus && npx tsc ... && npx vitest ...", so npx must resolve
# inside the sandbox; argv assertions cannot prove a bwrap-level mount
# actually works, only a real invocation can.
real_node = shutil.which("node")
if not real_node:
pytest.skip("no node on PATH to relocate for this canary")
real_prefix = Path(real_node).resolve().parent.parent
if not (real_prefix / "lib" / "node_modules" / "npm").is_dir():
pytest.skip(f"node at {real_node} has no npm under its install prefix")
toolcache = tmp_path / "toolcache" / "node" / "22.18.0" / "x64"
shutil.copytree(real_prefix, toolcache, symlinks=True)
relocated_node = toolcache / "bin" / "node"
assert relocated_node.exists()
real_which = shutil.which
monkeypatch.setattr(
"workflow_bench.proposer_sandbox.shutil.which",
lambda name: str(relocated_node) if name == "node" else real_which(name),
)
clone = tmp_path / "clone"
clone.mkdir()
with prepare_sandbox(clone=clone, claude_bin=Path(sys.executable), preflight=True) as sandbox:
result = sandbox.run(["/bin/sh", "-c", "command -v npx && npx --version"], timeout=60)
assert result.ok, result.stderr_tail
@pytest.mark.skipif(
os.environ.get("GITNEXUS_REQUIRE_BWRAP_CANARY") != "1",
reason="real Bubblewrap canary is mandatory in the named Ubuntu CI job",
@@ -1067,3 +750,4 @@ for line in sys.stdin:
assert bash_result.get("is_error") is not True, bash_result
assert (clone / "bash-called").read_text() == "canary"
assert (clone / "mcp-called").read_text() == "ok"
-119
View File
@@ -262,122 +262,3 @@ def test_phase_workspace_accepts_new_regular_review_output(tmp_path):
artifact.write_text("new review")
runner_artifacts.enforce_phase_workspace(tmp_path, before, allowed_artifact=artifact)
def test_phase_workspace_ignores_claude_sandbox_bootstrap_noise(tmp_path):
# Reproduced empirically: Claude Code's own enableWeakerNestedSandbox
# bootstrap creates this exact set of paths on every session regardless
# of task or model output (a trivial "say OK" prompt was enough). None
# of it is something the model decided to write, so it must not read as
# an unauthorized planning-phase change.
before = runner_artifacts.workspace_snapshot(tmp_path)
(tmp_path / ".claude" / "agents").mkdir(parents=True)
(tmp_path / ".claude" / "commands").mkdir(parents=True)
(tmp_path / ".claude" / ".cc-writes").write_text("{}")
(tmp_path / ".env").write_text("")
(tmp_path / ".env.development.local").write_text("")
(tmp_path / ".npmrc").write_text("")
(tmp_path / "package.json").write_text("{}")
(tmp_path / "node_modules").mkdir()
(tmp_path / "node_modules" / ".bin").mkdir()
artifact = tmp_path / "review-output.md"
artifact.write_text("new review")
runner_artifacts.enforce_phase_workspace(tmp_path, before, allowed_artifact=artifact)
def test_phase_workspace_still_rejects_a_genuinely_unauthorized_change(tmp_path):
# The bootstrap-noise exclusion must stay narrow: an actual source-file
# edit outside the allowed artifact still has to be caught.
before = runner_artifacts.workspace_snapshot(tmp_path)
(tmp_path / "src.py").write_text("changed")
artifact = tmp_path / "review-output.md"
artifact.write_text("new review")
with pytest.raises(ValueError, match="unauthorized workspace path"):
runner_artifacts.enforce_phase_workspace(tmp_path, before, allowed_artifact=artifact)
def test_phase_workspace_ignores_nested_claude_sandbox_bootstrap_noise(tmp_path):
# Claude Code bootstraps into whatever directory it is running in, not just
# the workspace root. The benchmark's task prompts cd into gitnexus/, so the
# same noise lands one level down -- observed verbatim in skill-evolution run
# 29861768554, where 13 of 18 sessions failed with
# "phase changed unauthorized workspace path(s): gitnexus/.claude/.cc-writes".
nested = tmp_path / "gitnexus" / ".claude"
nested.mkdir(parents=True)
(nested / "settings.local.json").write_text("{}")
before = runner_artifacts.workspace_snapshot(tmp_path)
(nested / ".cc-writes").write_text("{}")
artifact = tmp_path / "review-output.md"
artifact.write_text("new review")
runner_artifacts.enforce_phase_workspace(tmp_path, before, allowed_artifact=artifact)
def test_phase_workspace_does_not_descend_into_nested_bootstrap_directories(tmp_path):
# The exclusion must skip an entry before it is queued for traversal, so
# content created *inside* the ignored directory stays invisible too.
nested = tmp_path / "gitnexus" / ".claude" / ".cc-writes"
nested.mkdir(parents=True)
before = runner_artifacts.workspace_snapshot(tmp_path)
(nested / "pending.json").write_text('{"writes": 1}')
artifact = tmp_path / "review-output.md"
artifact.write_text("new review")
runner_artifacts.enforce_phase_workspace(tmp_path, before, allowed_artifact=artifact)
def test_phase_workspace_still_rejects_nested_real_claude_config(tmp_path):
# gitnexus/.claude/settings.local.json is real tracked repository content.
# Excluding ".claude" wholesale at depth would blind the check to it, so the
# exclusion must name only the entries Claude Code itself creates.
nested = tmp_path / "gitnexus" / ".claude"
nested.mkdir(parents=True)
settings = nested / "settings.local.json"
settings.write_text("{}")
before = runner_artifacts.workspace_snapshot(tmp_path)
settings.write_text('{"permissions": "changed"}')
artifact = tmp_path / "review-output.md"
artifact.write_text("new review")
with pytest.raises(ValueError, match="unauthorized workspace path"):
runner_artifacts.enforce_phase_workspace(tmp_path, before, allowed_artifact=artifact)
def test_phase_workspace_still_rejects_nested_package_json(tmp_path):
# package.json is in WORKSPACE_SNAPSHOT_BOOTSTRAP_NOISE, but only as a
# workspace-root entry: gitnexus/package.json is real tracked content whose
# edits must still be caught.
nested = tmp_path / "gitnexus"
nested.mkdir()
manifest = nested / "package.json"
manifest.write_text("{}")
before = runner_artifacts.workspace_snapshot(tmp_path)
manifest.write_text('{"version": "9.9.9"}')
artifact = tmp_path / "review-output.md"
artifact.write_text("new review")
with pytest.raises(ValueError, match="unauthorized workspace path"):
runner_artifacts.enforce_phase_workspace(tmp_path, before, allowed_artifact=artifact)
def test_phase_workspace_still_sees_writes_under_a_pre_existing_nested_claude_dir(tmp_path):
# Every excluded name is a blind spot. .claude/agents and .claude/commands
# are deliberately NOT excluded at depth: once a .claude directory exists
# (gitnexus/.claude/settings.local.json is tracked), anything written
# underneath an excluded entry is invisible to this check, and Claude Code
# loads .claude/agents relative to its cwd -- which these tasks point at
# gitnexus/. A planning phase must not be able to plant a definition there
# for the later work phase to read.
nested = tmp_path / "gitnexus" / ".claude"
nested.mkdir(parents=True)
(nested / "settings.local.json").write_text("{}")
before = runner_artifacts.workspace_snapshot(tmp_path)
(nested / "agents").mkdir()
(nested / "agents" / "planted.md").write_text("planted agent definition")
artifact = tmp_path / "review-output.md"
artifact.write_text("new review")
with pytest.raises(ValueError, match="unauthorized workspace path"):
runner_artifacts.enforce_phase_workspace(tmp_path, before, allowed_artifact=artifact)
+1 -55
View File
@@ -9,7 +9,7 @@ from pathlib import Path
import pytest
from workflow_bench.proposer_sandbox import VITE_TEMP_DIR, SandboxError
from workflow_bench.proposer_sandbox import SandboxError
from workflow_bench.oracle_assets import TaskOracleSnapshot
from workflow_bench.runner_tasks import resolve_task_bindings
from workflow_bench.task_assets import TaskAssetCache, stage_task_assets
@@ -113,27 +113,6 @@ def test_small_assets_use_a_bounded_buffered_fallback(monkeypatch, tmp_path: Pat
assert (clone / "second").read_bytes() == b"def"
def test_default_buffered_fallback_budget_covers_a_realistic_large_asset(
monkeypatch,
tmp_path: Path,
) -> None:
# 20 MiB exceeds the old 16 MiB default but must fit comfortably under
# the current default, proving the real (non-monkeypatched) budget
# constant is sized for a realistic large sandbox_copy asset such as the
# harness's own pre-built graph index, not just tiny fixtures.
payload = os.urandom(20 * 1024 * 1024)
repo, task = _repo_and_task(tmp_path, {"large": payload})
clone = tmp_path / "clone"
clone.mkdir()
monkeypatch.setattr(task_assets, "_try_reflink", lambda *_args: False)
with TaskAssetCache(tmp_path / "cache") as cache:
snapshot = cache.prepare(task, repo=repo, resolved_sha=SHA)
snapshot.materialize(clone)
assert (clone / "large").read_bytes() == payload
def test_large_asset_without_reflink_fails_before_publish_and_cleans_staging(
monkeypatch,
tmp_path: Path,
@@ -410,36 +389,3 @@ def test_resolved_task_binding_carries_dependency_digests_and_rejects_live_drift
(repo / "dependency" / "package.json").write_bytes(b'{"version":2}')
with pytest.raises(ValueError, match="definition drifted"):
resolve_task_bindings([task], [binding], oracle_snapshots=[oracle])
def test_node_modules_dependency_snapshot_captures_the_vite_temp_mount_point(tmp_path: Path) -> None:
# bwrap cannot mkdir a mount point inside an already-read-only bind, so the
# directory vite needs must exist in the captured dependency bytes. It is
# recorded during capture, which puts it inside the manifest and both
# dependency digests rather than leaving it an untracked mutation of a
# digest-bound snapshot.
repo, _ = _repo_and_task(tmp_path, {"dependency/package.json": b'{"version":1}'})
task = {
"sandbox_copy": [],
"sandbox_dependencies": [{"source": "dependency", "target": "gitnexus/node_modules"}],
}
with TaskAssetCache(tmp_path / "cache") as cache:
snapshot = cache.prepare(task, repo=repo, resolved_sha=SHA)
captured = {entry.path.as_posix() for entry in snapshot.dependencies[0].entries}
assert f"payload/{VITE_TEMP_DIR}" in captured
vite_temp = next((snapshot.root / "dependencies").glob(f"*/payload/{VITE_TEMP_DIR}"))
assert vite_temp.is_dir()
def test_non_node_modules_dependency_snapshot_has_no_vite_temp(tmp_path: Path) -> None:
# The capture is scoped to dependency mounts whose target is node_modules;
# an unrelated vendored dependency is captured byte-for-byte as declared.
repo, _ = _repo_and_task(tmp_path, {"dependency/package.json": b'{"version":1}'})
task = {
"sandbox_copy": [],
"sandbox_dependencies": [{"source": "dependency", "target": "vendor/dependency"}],
}
with TaskAssetCache(tmp_path / "cache") as cache:
snapshot = cache.prepare(task, repo=repo, resolved_sha=SHA)
captured = {entry.path.as_posix() for entry in snapshot.dependencies[0].entries}
assert not any(path.endswith(VITE_TEMP_DIR) for path in captured)
+1 -58
View File
@@ -10,7 +10,6 @@ import yaml
from workflow_bench.runner import (
aggregate,
broken_incumbent_arms,
build_parser,
infra_error_record,
normalized_model_identifier,
@@ -65,7 +64,6 @@ def test_aggregate_takes_medians_and_counts_resolved():
"valid_runs": 3,
"excluded_runs": 0,
"transcripts_missing": 0,
"error_kinds": {},
}
@@ -174,7 +172,7 @@ def test_eval_ci_uses_locked_uv_and_blocking_native_containment_jobs():
}
assert containment["timeout-minutes"] == 20
assert containment_node_setup["with"] == {
"node-version": "22.18.0",
"node-version": "22.16.0",
"cache": "npm",
"cache-dependency-path": "gitnexus/package-lock.json\ngitnexus-shared/package-lock.json\n",
}
@@ -335,61 +333,6 @@ def test_render_report_surfaces_excluded_and_unverified_runs():
assert "no locatable session transcript" in report
def test_render_report_surfaces_why_each_row_failed():
results = {
"t": {
"workflow": aggregate(
[record(resolved=False, error_kind="plan-evidence-invalid")],
),
}
}
report = render_report(results)
assert "plan-evidence-invalid×1" in report
def test_broken_incumbent_arms_flags_an_incumbent_that_resolved_nothing():
results = {
"t1": {"workflow": aggregate([record(resolved=False, error_kind="plan-evidence-invalid")])},
"t2": {"workflow": aggregate([record(resolved=False, error_kind="plan-evidence-invalid")])},
}
assert broken_incumbent_arms(results, {"workflow"}) == ["workflow"]
def test_broken_incumbent_arms_ignores_a_merely_underperforming_candidate():
# The incumbent works fine; only the candidate arm fails. That's a normal,
# expected "bad candidate" outcome and must not read as a broken harness.
results = {
"t1": {
"workflow": aggregate([record(resolved=True)]),
"candidate_workflow": aggregate([record(resolved=False, error_kind="verify-failed")]),
},
}
assert broken_incumbent_arms(results, {"workflow"}) == []
def test_broken_incumbent_arms_flags_an_incumbent_with_zero_valid_runs():
# Every run excluded via an excluded-but-non-systemic error_kind
# ("evidence-unverified"): valid_runs == 0 for every task, which the old
# `valid_runs > 0` guard let sail through silently, and which the outage
# streak breaker also doesn't catch (it resets rather than accumulates
# on this exact error_kind -- see test_systemic_outage_streak_resets_on_non_outage).
results = {
"t1": {"workflow": aggregate([record(resolved=False, error_kind="evidence-unverified")])},
"t2": {"workflow": aggregate([record(resolved=False, error_kind="evidence-unverified")])},
}
assert results["t1"]["workflow"]["valid_runs"] == 0
assert broken_incumbent_arms(results, {"workflow"}) == ["workflow"]
def test_broken_incumbent_arms_ignores_partial_incumbent_failure():
# Resolved in at least one task — struggling, not broken.
results = {
"t1": {"workflow": aggregate([record(resolved=False, error_kind="verify-failed")])},
"t2": {"workflow": aggregate([record(resolved=True)])},
}
assert broken_incumbent_arms(results, {"workflow"}) == []
def test_infra_error_record_captures_the_failure_and_is_excluded():
exc = subprocess.TimeoutExpired(cmd="claude -p", timeout=5)
rec = infra_error_record(exc)
+2 -81
View File
@@ -167,56 +167,6 @@ def test_run_claude_forwards_the_named_model_to_every_session(monkeypatch, tmp_p
assert captured[captured.index("--model") + 1] == "claude-sonnet-4-20250514"
def test_run_claude_restricts_tools_via_tools_flag_outside_bare(monkeypatch, tmp_path):
# Outside --bare, the built-in toolset defaults to everything (subagents,
# WebFetch, Task, ...) and --allowedTools only pre-approves within that —
# it does not narrow it. --tools is what actually restricts the set, so a
# non-bare arm session must pass it or it silently gets a far wider
# toolset than intended.
captured: list[str] = []
def fake_run(command, **kwargs):
captured.extend(command)
return fake_cli_result(VALID_REPORT)
monkeypatch.setattr(runner_sessions, "run_managed", fake_run)
runner.run_claude(
"task",
tmp_path,
claude_bin="claude",
timeout=5,
bare=False,
allowed_tools=["Read", "Edit", "Bash", "Skill"],
)
tools_idx = captured.index("--tools")
assert captured[tools_idx + 1 : tools_idx + 5] == ["Read", "Edit", "Bash", "Skill"]
allowed_idx = captured.index("--allowedTools")
assert captured[allowed_idx + 1 : allowed_idx + 5] == ["Read", "Edit", "Bash", "Skill"]
def test_run_claude_omits_tools_flag_under_bare(monkeypatch, tmp_path):
# --bare already hard-restricts to Bash/Edit/Read on its own (a Claude
# Code design choice, not something --tools/--allowedTools can widen or
# narrow further), so bare sessions must not also pass --tools.
captured: list[str] = []
def fake_run(command, **kwargs):
captured.extend(command)
return fake_cli_result(VALID_REPORT)
monkeypatch.setattr(runner_sessions, "run_managed", fake_run)
runner.run_claude(
"task",
tmp_path,
claude_bin="claude",
timeout=5,
bare=True,
allowed_tools=["Read", "Edit", "Bash", "Skill"],
)
assert "--tools" not in captured
assert "--allowedTools" in captured
@pytest.mark.parametrize(
("proc", "expected_kind"),
[
@@ -332,15 +282,6 @@ def test_agent_tool_grants_are_exact_and_nomcp_has_no_graph_tools(monkeypatch, t
assert captured[3]["mcp_config_json"] == '{"mcpServers":{}}'
assert captured[3]["disallowed_tools"] == ["Skill", "mcp__gitnexus"]
# --bare hard-disables the Skill tool and every mcp__* tool regardless of
# --allowedTools (a Claude Code design choice, not something the harness
# can override) -- every arm here except baseline_nomcp needs Skill
# and/or MCP tools, so only baseline_nomcp may still run under --bare.
assert captured[0]["bare"] is False # workflow: planning session
assert captured[1]["bare"] is False # review
assert captured[2]["bare"] is False # workflow_direct
assert captured[3]["bare"] is True # baseline_nomcp
def test_mcp_config_uses_only_the_minimal_pinned_harness_runtime(monkeypatch, tmp_path):
runtime = tmp_path / "gitnexus"
@@ -349,12 +290,10 @@ def test_mcp_config_uses_only_the_minimal_pinned_harness_runtime(monkeypatch, tm
runtime / "dist" / "cli",
runtime / "node_modules",
runtime / "vendor",
runtime / "hooks" / "claude",
shared / "dist",
):
directory.mkdir(parents=True)
(runtime / "dist" / "cli" / "index.js").write_text("")
(runtime / "hooks" / "claude" / "resolve-analyze-cmd.cjs").write_text("")
(runtime / "package.json").write_text(json.dumps({"version": runner.PINNED_GITNEXUS_VERSION}))
(runtime / "node_modules" / "gitnexus-shared").symlink_to(shared, target_is_directory=True)
(shared / "package.json").write_text(json.dumps({"name": "gitnexus-shared"}))
@@ -377,7 +316,6 @@ def test_mcp_config_uses_only_the_minimal_pinned_harness_runtime(monkeypatch, tm
(runtime / "vendor", f"{runner.SANDBOX_GITNEXUS}/vendor"),
(shared / "dist", f"{runner.SANDBOX_GITNEXUS_SHARED}/dist"),
(shared / "package.json", f"{runner.SANDBOX_GITNEXUS_SHARED}/package.json"),
(runtime / "hooks" / "claude", f"{runner.SANDBOX_GITNEXUS}/hooks/claude"),
]
package = json.loads((runtime / "package.json").read_text())
assert package["version"] == runner.PINNED_GITNEXUS_VERSION
@@ -392,12 +330,6 @@ def test_mcp_config_uses_only_the_minimal_pinned_harness_runtime(monkeypatch, tm
assert shared / forbidden not in mounted_sources
assert f"{runner.SANDBOX_GITNEXUS_SHARED}/{forbidden}" not in mounted_targets
# Only hooks/claude is exposed, not the whole hooks/ directory (which also
# has an unrelated hooks/antigravity/ tree) and not the runtime root itself.
assert runtime / "hooks" not in mounted_sources
assert runtime / "hooks" / "antigravity" not in mounted_sources
assert f"{runner.SANDBOX_GITNEXUS}/hooks" not in mounted_targets
@pytest.mark.skipif(
os.environ.get("GITNEXUS_REQUIRE_BWRAP_CANARY") != "1",
@@ -415,7 +347,6 @@ def test_real_bubblewrap_runtime_mount_imports_cli_without_exposing_checkout(tmp
f"{runner.SANDBOX_GITNEXUS}/vendor",
f"{runner.SANDBOX_GITNEXUS_SHARED}/dist/index.js",
f"{runner.SANDBOX_GITNEXUS_SHARED}/package.json",
f"{runner.SANDBOX_GITNEXUS}/hooks/claude/resolve-analyze-cmd.cjs",
]
forbidden = [
f"{runner.SANDBOX_GITNEXUS}/{relative}"
@@ -438,26 +369,16 @@ def test_real_bubblewrap_runtime_mount_imports_cli_without_exposing_checkout(tmp
preflight=True,
) as sandbox:
visibility = sandbox.run(
[runner.SANDBOX_NODE, "-e", visibility_script],
["/usr/local/bin/node", "-e", visibility_script],
timeout=10,
)
imported = sandbox.run(
[runner.SANDBOX_NODE, runner.SANDBOX_GITNEXUS_ENTRYPOINT, "--version"],
timeout=10,
)
# --version never reaches the `analyze` command, which is loaded via a
# lazy dynamic import and is the only path that pulls in
# resolve-invocation.ts's module-load-time require of hooks/claude/
# resolve-analyze-cmd.cjs. Require the compiled analyze module
# directly so this canary actually exercises that chain.
analyze_imported = sandbox.run(
[runner.SANDBOX_NODE, "-e", f"require('{runner.SANDBOX_GITNEXUS}/dist/cli/analyze.js')"],
["/usr/local/bin/node", runner.SANDBOX_GITNEXUS_ENTRYPOINT, "--version"],
timeout=10,
)
assert visibility.ok, visibility.stderr_tail
assert imported.ok, imported.stderr_tail
assert analyze_imported.ok, analyze_imported.stderr_tail
assert imported.stdout_tail.strip() == runner.PINNED_GITNEXUS_VERSION
+2 -95
View File
@@ -26,18 +26,7 @@ SANDBOX_HOME = "/home/agent"
SANDBOX_TMP = "/tmp"
SANDBOX_CLAUDE = "/opt/claude/claude"
SANDBOX_SHELL_PREFIX = "/opt/claude/shell-prefix"
SANDBOX_PYTHON3 = "/opt/claude/python3"
SANDBOX_NODE = "/opt/claude/node"
SANDBOX_NODE_PREFIX = "/opt/claude/nodejs"
# Vite transpiles a TypeScript config into <node_modules>/.vite-temp before it
# loads anything, so a read-only dependency mount makes `vitest` die with EROFS
# before a single test runs -- and every task verify command and every hidden
# oracle ends in `npx vitest run <test>`. bwrap cannot create a mount point
# inside an already-read-only bind, so the directory is captured into the
# dependency snapshot (task_assets.py) and a tmpfs is overlaid on it here.
VITE_TEMP_DIR = ".vite-temp"
DEPENDENCY_MOUNT_BASENAME = "node_modules"
SANDBOX_PATH = f"/opt/claude:{SANDBOX_NODE_PREFIX}/bin:/usr/local/bin:/usr/bin:/bin"
SANDBOX_PATH = "/opt/claude:/usr/local/bin:/usr/bin:/bin"
SANDBOX_GITNEXUS = "/opt/gitnexus"
SANDBOX_GITNEXUS_SHARED = "/opt/gitnexus-shared"
SANDBOX_GITNEXUS_REGISTRY = "/opt/gitnexus-registry"
@@ -360,58 +349,10 @@ def build_claude_settings() -> str:
def _runtime_mount_args() -> list[str]:
args: list[str] = []
system_trees = ("/usr", "/bin", "/lib", "/lib64")
for raw in system_trees:
for raw in ("/usr", "/bin", "/lib", "/lib64"):
path = Path(raw)
if path.exists():
args += ["--ro-bind", raw, raw]
# sanitized_graph.py and runner_sessions.py invoke the sandboxed graph
# CLI via SANDBOX_NODE. Bind whatever `node` actually resolves to on PATH
# there -- true node location varies by host (GitHub-hosted runner images
# happen to have one under /usr/local/bin; a self-hosted runner's
# actions/setup-node installs into its own tool-cache directory instead).
# Target must be a fresh path like /opt/claude/... rather than anywhere
# under /usr, /bin, /lib, or /lib64: those are already read-only bound
# above, and bwrap can't create a new mount-point file inside an
# already-read-only tree when the real path doesn't already exist there
# (the exact case a self-hosted runner hits, and the reason this bind
# exists at all).
node_bin = shutil.which("node")
if node_bin:
args += ["--ro-bind", node_bin, SANDBOX_NODE]
# The single-binary bind above gives SANDBOX_NODE but NOT npm or npx:
# those are symlinks into ../lib/node_modules/npm/bin/*-cli.js, so the
# install prefix carrying both bin/ and lib/node_modules has to be
# mounted for them to resolve at all. When node really lives under a
# system tree (/usr/local/bin on GitHub-hosted images) the prefix is
# already inside the wholesale read-only binds above and npm/npx came
# along for free -- which is exactly why this gap stayed invisible
# until a self-hosted runner put node in actions/setup-node's tool
# cache, outside /usr, and every task verify command
# ("cd gitnexus && npx tsc ... && npx vitest ...") died with
# "/bin/sh: 1: npx: not found". Skip the redundant bind in the
# already-covered case so the mount surface stays minimal.
#
# The prefix is only ever derived from a real <prefix>/bin/node layout
# that actually carries npm. Deriving it as parent.parent unconditionally
# would mount an unrelated ancestor whenever node sits somewhere else:
# /opt/bin/node would bind all of /opt (every tool cache on a hosted
# runner) and a bare <dir>/node would bind <dir>'s parent. This function
# exists to keep the sandbox surface minimal, so an unrecognized layout
# binds nothing extra and simply leaves npx unavailable, exactly as
# before.
node_bin_dir = Path(node_bin).resolve().parent
node_prefix = node_bin_dir.parent
# Test the property actually needed -- a working npx next to node in a
# real bin/ directory -- rather than a proxy like lib/node_modules/npm.
# .exists() follows the symlink, so a dangling npx correctly fails: it
# would not survive the mount either. Requiring the "bin" name keeps
# the parent.parent derivation honest; an npx sitting directly beside
# node in a flat directory would make that derivation name the wrong
# prefix.
provides_npx = node_bin_dir.name == "bin" and (node_bin_dir / "npx").exists()
if provides_npx and not any(node_prefix.is_relative_to(tree) for tree in system_trees):
args += ["--ro-bind", str(node_prefix), SANDBOX_NODE_PREFIX]
for raw in (
"/etc/ssl",
"/etc/hosts",
@@ -443,24 +384,6 @@ def _create_shell_prefix_wrapper(private_root: Path) -> Path:
return wrapper
def _create_python3_wrapper(private_root: Path) -> Path:
"""A trusted, self-owned Python 3 launcher for evidence-provenance.mjs's atomic mover.
/usr/bin/python3 is a real system binary, but it's root-owned on the host.
Inside this --unshare-user sandbox only the calling uid is mapped (root is
not), so root-owned files surface as the kernel's overflow uid — which
evidence-provenance.mjs's PATH-scan correctly refuses to trust. This
wrapper is freshly created by the same host process that owns
home/temp/shell-prefix, so it maps to the sandbox's own trusted uid
instead, and simply execs the real interpreter through to do the work.
"""
wrapper = private_root / "python3"
wrapper.write_text('#!/bin/bash\nset -eu\nexec /usr/bin/python3 "$@"\n')
wrapper.chmod(0o500)
return wrapper
def _resolve_executable(executable: Path | str | None, default: str) -> Path:
raw = os.fspath(executable) if executable is not None else shutil.which(default)
if not raw:
@@ -686,20 +609,6 @@ def _sandbox_command_prefix(
]
for mount in mounts:
args += ["--ro-bind", str(mount.source), mount.target]
# Overlay an empty writable tmpfs on the one path vite must write.
# Everything else in the mount, and the whole workspace, stays
# read-only, and the overlay lives only inside the sandbox -- it never
# reaches the host clone the credited patch is captured from.
#
# Gate on the mount SOURCE actually containing the directory, not on
# the target name: bwrap cannot create a mount point inside an
# already-read-only bind, so a tmpfs can only be overlaid where the
# directory already exists in the bound bytes. task_assets.py captures
# it into dependency-snapshot node_modules; other node_modules mounts
# (e.g. the trusted GitNexus runtime at /opt/gitnexus/node_modules) do
# not carry it, and overlaying them would fail with EROFS.
if PurePosixPath(mount.target).name == DEPENDENCY_MOUNT_BASENAME and (mount.source / VITE_TEMP_DIR).is_dir():
args += ["--tmpfs", f"{mount.target}/{VITE_TEMP_DIR}"]
args += ["--chdir", SANDBOX_WORKSPACE, "--"]
return args
@@ -732,7 +641,6 @@ def prepare_sandbox(
directory.mkdir(mode=0o700)
directory.chmod(0o700)
shell_prefix = _create_shell_prefix_wrapper(private_root)
python3_wrapper = _create_python3_wrapper(private_root)
# Claude may discover user-level skills below HOME. Keep the rest of HOME
# writable for normal CLI state, but overlay an immutable empty skills root
# so a model cannot shadow the evaluated repository/plugin skill by name.
@@ -743,7 +651,6 @@ def prepare_sandbox(
*read_only_mounts,
ReadOnlyMount(source=user_skills, target=SANDBOX_USER_SKILLS),
ReadOnlyMount(source=shell_prefix, target=SANDBOX_SHELL_PREFIX),
ReadOnlyMount(source=python3_wrapper, target=SANDBOX_PYTHON3),
)
primary: BaseException | None = None
try:
+6 -62
View File
@@ -78,7 +78,6 @@ from .proposer_sandbox import (
SANDBOX_GITNEXUS as SANDBOX_GITNEXUS,
SANDBOX_GITNEXUS_REGISTRY,
SANDBOX_GITNEXUS_SHARED as SANDBOX_GITNEXUS_SHARED,
SANDBOX_NODE as SANDBOX_NODE,
SANDBOX_WORKSPACE,
ReadOnlyMount,
SandboxError,
@@ -403,13 +402,6 @@ def run_arm(
auth_token=args.auth_token,
base_url=args.base_url,
)
# --bare hard-disables the Skill tool and every mcp__* tool — by Claude
# Code design, not a bug (--allowedTools can't restore what --bare
# removes). Every arm except baseline_nomcp needs Skill and/or MCP tools,
# so only baseline_nomcp can keep --bare's tighter isolation; the rest
# rely on ANTHROPIC_API_KEY alone (the sandboxed HOME has no OAuth/
# keychain state to conflict with it).
bare = arm == "baseline_nomcp"
common = {
"claude_bin": sandbox.claude_bin,
"timeout": args.timeout,
@@ -420,7 +412,7 @@ def run_arm(
read_only_paths=_evaluated_skill_roots(worktree, arm),
),
"require_pid_namespace": True,
"bare": bare,
"bare": True,
"settings_json": sandbox.settings_json,
"strict_mcp_config": True,
"mcp_config_json": sandbox_mcp_config(),
@@ -680,21 +672,13 @@ def aggregate(records: list[dict[str, Any]]) -> dict[str, Any]:
# unmeasured run makes the whole median unavailable so the gate won't rank
# a candidate on a cost that was never actually captured.
valid_costs = [r.get("cost_usd") for r in valid]
out["cost_usd"] = (
None if (not valid or any(cost is None for cost in valid_costs)) else statistics.median(valid_costs)
)
out["cost_usd"] = None if (not valid or any(cost is None for cost in valid_costs)) else statistics.median(valid_costs)
out["resolved"] = sum(1 for r in records if r["resolved"])
out["runs"] = len(records)
out["valid_runs"] = len(valid)
out["excluded_runs"] = len(records) - len(valid)
out["transcripts_missing"] = sum(1 for r in records if r.get("transcript_missing"))
out["class"] = records[0].get("class", "")
error_kinds: dict[str, int] = {}
for r in records:
kind = r.get("error_kind")
if kind:
error_kinds[kind] = error_kinds.get(kind, 0) + 1
out["error_kinds"] = error_kinds
return out
@@ -711,33 +695,6 @@ def savings(baseline: dict[str, Any], workflow: dict[str, Any]) -> dict[str, Any
return out
def broken_incumbent_arms(
results: dict[str, dict[str, dict[str, Any]]],
incumbent_arms: set[str],
) -> list[str]:
"""Incumbent arms that resolved nothing across every task they ran.
An incumbent arm is the currently-shipped, presumably-working skill: if it
resolves NOTHING across every task it ran, that reads as an environment or
harness failure (missing trusted interpreter, stale skill fingerprint,
sandbox misconfiguration), not a skill regression. A candidate merely
underperforming is a normal, expected outcome and must not trip this —
only checking incumbents keeps that distinction.
Deliberately does NOT require valid_runs > 0 per task: an incumbent that
fails every run with an excluded-but-non-systemic error_kind (e.g.
"evidence-unverified", which the outage-streak breaker explicitly resets
on rather than accumulates) would otherwise never accumulate a single
valid run and sail through silently — the exact "quiet no-promotion"
outcome this guard exists to catch, and arguably worse than the
some-runs-resolved-zero case since here nothing completed at all.
aggregate() never marks an excluded/unverifiable row resolved=True, so
resolved == 0 alone already covers both cases.
"""
present = incumbent_arms & {arm for arms in results.values() for arm in arms}
return sorted(arm for arm in present if all(arms[arm]["resolved"] == 0 for arms in results.values() if arm in arms))
def _na(value: Any) -> Any:
"""Render an unmeasured metric as ``n/a`` instead of a misleading number."""
return "n/a" if value is None else value
@@ -762,8 +719,8 @@ def render_report(results: dict[str, dict[str, dict[str, Any]]]) -> str:
"efficiency, sum usage from the session transcripts instead",
"(dedup events sharing one message.id).",
"",
"| task | class | arm | resolved | input | cache_create | cache_read | output | cost $ | wall s | turns | churn | errors |",
"| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |",
"| task | class | arm | resolved | input | cache_create | cache_read | output | cost $ | wall s | turns | churn |",
"| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |",
]
for task_id, arms in results.items():
for arm, agg in arms.items():
@@ -771,14 +728,12 @@ def render_report(results: dict[str, dict[str, dict[str, Any]]]) -> str:
resolved_cell = f"{agg['resolved']}/{agg.get('valid_runs', agg['runs'])}"
if excluded:
resolved_cell += f" ({excluded} excluded)"
error_cell = ", ".join(f"{kind}×{count}" for kind, count in sorted(agg.get("error_kinds", {}).items()))
lines.append(
f"| {task_id} | {agg['class']} | {arm} | {resolved_cell} "
f"| {agg['input_tokens']:.0f} | {agg['cache_creation_input_tokens']:.0f} "
f"| {agg['cache_read_input_tokens']:.0f} | {agg['output_tokens']:.0f} "
f"| {_cost_cell(agg['cost_usd'])} | {agg['duration_s']:.0f} | {agg['num_turns']:.0f} "
f"| {agg['diff_files']:.0f}/+{agg['diff_insertions']:.0f}/−{agg['diff_deletions']:.0f} "
f"| {error_cell} |"
f"| {agg['diff_files']:.0f}/+{agg['diff_insertions']:.0f}/−{agg['diff_deletions']:.0f} |"
)
for arm in arms:
if arm != "baseline" and "baseline" in arms:
@@ -787,7 +742,7 @@ def render_report(results: dict[str, dict[str, dict[str, Any]]]) -> str:
f"| {task_id} | {arms[arm]['class']} | **{arm} savings %** | — "
f"| {s['input_tokens']} | {s['cache_creation_input_tokens']} "
f"| {s['cache_read_input_tokens']} | {s['output_tokens']} "
f"| {_na(s['cost_usd'])} | {s['duration_s']} | — | — | — |"
f"| {_na(s['cost_usd'])} | {s['duration_s']} | — | — |"
)
lines.append("")
all_aggs = [agg for arms in results.values() for agg in arms.values()]
@@ -1371,17 +1326,6 @@ def main() -> None:
}
(out_dir / "promotion.json").write_text(json.dumps(promotion, indent=2) + "\n")
print(f"\n{report}\n\nWritten to {out_dir}/")
broken_incumbents = broken_incumbent_arms(results, set(CANDIDATE_ARMS.values()))
if broken_incumbents:
# Fail loudly rather than let a broken environment read as a quiet
# "no promotion, incumbent stands."
print(
f"[harness-health] incumbent arm(s) {', '.join(broken_incumbents)} resolved zero "
"tasks across every valid run — this looks like an environment/harness failure, "
"not a normal candidate miss. See the errors column in report.md and error_detail "
"in results.jsonl. Exiting non-zero rather than reporting a quiet no-promotion."
)
raise SystemExit(1)
if outage_tripped:
# Non-zero exit so a driver (evolve.py) treats the partial benchmark as a
# failed run and halts instead of proposing from outage-truncated evidence.
+2 -71
View File
@@ -21,64 +21,6 @@ MAX_WORKSPACE_SNAPSHOT_ENTRIES = 100_000
MAX_WORKSPACE_SNAPSHOT_PATH_BYTES = 16 * 1024 * 1024
MAX_WORKSPACE_SNAPSHOT_FILE_BYTES = 1024 * 1024 * 1024
# Claude Code's own enableWeakerNestedSandbox bootstrap creates these paths on
# EVERY session regardless of task or model output -- reproduced empirically
# with a trivial "say OK" prompt: a synthetic package.json/lockfiles/
# node_modules, a full set of .env variants, and .claude/agents,
# .claude/commands, .claude/.cc-writes. None of this is something the model
# decided to write, so it must not count as an "unauthorized" workspace
# change during the planning-phase boundary check (the one thing this
# snapshot is used for -- see workspace_snapshot's callers). Mirrors the
# pre-existing .git exclusion below, which is the same kind of harness/tool
# noise rather than substantive diff.
WORKSPACE_SNAPSHOT_BOOTSTRAP_NOISE = frozenset(
{
".claude",
".env",
".env.development",
".env.development.local",
".env.local",
".env.production",
".env.production.local",
".env.test",
".env.test.local",
".gitmodules",
".npmrc",
".yarnrc",
".yarnrc.yml",
"bunfig.toml",
"node_modules",
"package-lock.json",
"package.json",
"pnpm-lock.yaml",
"yarn.lock",
}
)
# The set above is matched at the workspace ROOT only, because most of its
# entries (package.json, node_modules, the .env family) are also legitimate
# repository content further down the tree -- gitnexus/package.json and
# gitnexus/.claude/settings.local.json are both tracked files whose edits must
# still be caught. But Claude Code bootstraps into whatever directory it is
# running in, so a task whose prompt cd's into a subdirectory gets the same
# noise one level down. Observed in skill-evolution run 29861768554: 13 of 18
# sessions failed with "phase changed unauthorized workspace path(s):
# gitnexus/.claude/.cc-writes". That entry is matched at ANY depth -- never
# ".claude" itself, which holds real configuration.
#
# Deliberately only .cc-writes. Every excluded name is a blind spot: once a
# .claude directory already exists (gitnexus/.claude/settings.local.json is
# tracked), anything a phase writes underneath an excluded entry becomes
# invisible to this check, and Claude Code loads .claude/agents relative to
# its cwd -- which these tasks point at gitnexus/. Adding "agents" and
# "commands" here on the theory that they might also appear nested would let a
# planning phase plant a definition that the later work phase reads, with no
# evidence in the boundary check. Only .cc-writes was ever observed nested, so
# only .cc-writes is excluded; extend this set from an observed failure, never
# pre-emptively.
CLAUDE_BOOTSTRAP_DIR = ".claude"
CLAUDE_BOOTSTRAP_ENTRIES = frozenset({".cc-writes"})
IMPLEMENTATION_ARMS = frozenset(
{
"workflow",
@@ -110,19 +52,8 @@ class VerificationResult:
yield self.output
def _is_bootstrap_noise(relative: PurePosixPath) -> bool:
"""Report whether a walked entry is harness noise rather than workspace change."""
parts = relative.parts
if parts[0] == ".git" or parts[0] in WORKSPACE_SNAPSHOT_BOOTSTRAP_NOISE:
return True
return len(parts) >= 2 and parts[-2] == CLAUDE_BOOTSTRAP_DIR and parts[-1] in CLAUDE_BOOTSTRAP_ENTRIES
def workspace_snapshot(worktree: Path) -> dict[str, str]:
"""Hash the workspace without following links, excluding Git internals
and Claude Code's own sandbox-bootstrap noise (see
WORKSPACE_SNAPSHOT_BOOTSTRAP_NOISE)."""
"""Hash the workspace without following links, excluding Git internals."""
root = worktree.expanduser().absolute()
mode = root.lstat().st_mode
@@ -143,7 +74,7 @@ def workspace_snapshot(worktree: Path) -> dict[str, str]:
raise ValueError(f"workspace snapshot directory is unreadable: {directory}: {exc}") from exc
for entry in children:
relative = relative_dir / entry.name
if _is_bootstrap_noise(relative):
if relative.parts[0] == ".git":
continue
entry_count += 1
path_bytes += len(relative.as_posix().encode())
+1 -11
View File
@@ -18,7 +18,6 @@ from .proposer_sandbox import (
SANDBOX_GITNEXUS,
SANDBOX_GITNEXUS_REGISTRY,
SANDBOX_HOME,
SANDBOX_NODE,
SANDBOX_TMP,
SANDBOX_WORKSPACE,
SandboxError,
@@ -51,8 +50,6 @@ def measured_cost(raw: Any) -> float | None:
if not math.isfinite(raw) or raw < 0:
return None
return float(raw)
SANDBOX_GITNEXUS_ENTRYPOINT = f"{SANDBOX_GITNEXUS}/dist/cli/index.js"
SENSITIVE_EVENT_KEYS = frozenset(
{
@@ -116,7 +113,7 @@ def sandbox_mcp_config() -> str:
"PATH=/usr/local/bin:/usr/bin:/bin",
"LANG=C.UTF-8",
"GIT_TERMINAL_PROMPT=0",
SANDBOX_NODE,
"/usr/local/bin/node",
SANDBOX_GITNEXUS_ENTRYPOINT,
"mcp",
],
@@ -380,13 +377,6 @@ def run_claude(
if strict_mcp_config:
cmd += ["--strict-mcp-config", "--mcp-config", mcp_config_json or '{"mcpServers":{}}']
if allowed_tools:
# --bare's own hard-coded Bash/Edit/Read ceiling already scopes bare
# sessions; outside --bare the built-in toolset defaults to
# everything (subagents, WebFetch, Task, ...), so --tools is needed
# to actually restrict it — --allowedTools only pre-approves within
# whatever set is available, it does not narrow that set.
if not bare:
cmd += ["--tools", *allowed_tools]
cmd += ["--allowedTools", *allowed_tools]
if disable_slash_commands:
cmd.append("--disable-slash-commands")
-6
View File
@@ -217,12 +217,6 @@ def trusted_gitnexus_runtime_mounts() -> tuple[ReadOnlyMount, ...]:
f"{SANDBOX_GITNEXUS_SHARED}/package.json",
directory=False,
),
_validated_runtime_component(
runtime,
"hooks/claude",
f"{SANDBOX_GITNEXUS}/hooks/claude",
directory=True,
),
)
entrypoint = mounts[0].source / "cli" / "index.js"
+1 -2
View File
@@ -16,7 +16,6 @@ from .process_control import ManagedProcessError, run_managed
from .proposer_sandbox import (
SANDBOX_GITNEXUS,
SANDBOX_HOME,
SANDBOX_NODE,
SANDBOX_WORKSPACE,
ReadOnlyMount,
SandboxError,
@@ -249,7 +248,7 @@ def _run_graph_cli(
) -> bytes | None:
command = [
*prefix,
SANDBOX_NODE,
"/usr/local/bin/node",
SANDBOX_GITNEXUS_ENTRYPOINT,
*arguments,
]
+21 -51
View File
@@ -25,9 +25,7 @@ from pathlib import Path, PurePosixPath
from typing import Any
from .proposer_sandbox import (
DEPENDENCY_MOUNT_BASENAME,
SANDBOX_WORKSPACE,
VITE_TEMP_DIR,
ReadOnlyMount,
SandboxError,
_prepare_clone_target,
@@ -41,14 +39,9 @@ MAX_TASK_ASSET_ENTRIES = 100_000
MAX_TASK_ASSET_PATH_BYTES = 4_096
MAX_TASK_ASSET_BYTES = 2 * 1024 * 1024 * 1024
# The largest known real sandbox_copy asset in this harness is the shipped
# index above (~428 MiB estimated, ~290 MiB measured); budget comfortably
# above that so it can still materialize via buffered copy on a filesystem
# that cannot reflink (ext4 CI runners, 9p-backed dev mounts), while staying
# well below MAX_TASK_ASSET_BYTES so a genuinely oversized or malformed
# declaration still fails closed instead of silently paying for a slow full
# copy.
MAX_BUFFERED_FALLBACK_BYTES = 512 * 1024 * 1024
# A filesystem without reflink support may still run tiny fixtures. Large
# assets fail closed instead of silently returning to one full copy per arm.
MAX_BUFFERED_FALLBACK_BYTES = 16 * 1024 * 1024
COPY_CHUNK_BYTES = 1024 * 1024
# linux/fs.h: #define FICLONE _IOW(0x94, 9, int)
@@ -162,11 +155,9 @@ class TaskAssetSnapshot:
source = snapshot_root / Path(*dependency.snapshot_path.parts)
metadata = source.lstat()
expected_directory = dependency.kind == "directory"
if (
stat.S_ISLNK(metadata.st_mode)
or (expected_directory and not stat.S_ISDIR(metadata.st_mode))
or (not expected_directory and not stat.S_ISREG(metadata.st_mode))
):
if stat.S_ISLNK(metadata.st_mode) or (
expected_directory and not stat.S_ISDIR(metadata.st_mode)
) or (not expected_directory and not stat.S_ISREG(metadata.st_mode)):
raise SandboxError(f"dependency snapshot changed: {dependency.source}")
target = PurePosixPath(dependency.target)
_prepare_clone_target(
@@ -217,7 +208,9 @@ class TaskAssetCache:
repo_identity = _real_directory(repo, label="task asset repository")
declarations, relative_paths = _sandbox_copy_declarations(task)
dependency_declarations = _sandbox_dependency_declarations(task)
dependency_identity = tuple((declaration.source, declaration.target) for declaration in dependency_declarations)
dependency_identity = tuple(
(declaration.source, declaration.target) for declaration in dependency_declarations
)
definition = (str(repo_identity), resolved_sha, declarations, dependency_identity)
existing = self._by_definition.get(definition)
if existing is not None:
@@ -260,21 +253,6 @@ class TaskAssetCache:
dependency_builder.copy_descriptor(descriptor, PurePosixPath("payload"))
finally:
os.close(descriptor)
# vitest cannot start against a read-only node_modules: vite
# writes <node_modules>/.vite-temp/<config>.timestamp-*.mjs
# before loading a TypeScript config. bwrap cannot create
# that mount point inside an already-read-only bind, so the
# empty directory is captured here -- before the manifest and
# both dependency digests are computed, so it is part of the
# snapshot rather than an untracked mutation of it. The
# sandbox overlays a tmpfs on it; see VITE_TEMP_DIR.
payload_entry = dependency_builder.entries.get(PurePosixPath("payload"))
if (
payload_entry is not None
and payload_entry.kind == "directory"
and PurePosixPath(declaration.target).name == DEPENDENCY_MOUNT_BASENAME
):
dependency_builder.ensure_directory(PurePosixPath("payload") / VITE_TEMP_DIR)
dependency_entries = dependency_builder.finished_entries()
_validate_dependency_symlinks(
container,
@@ -479,14 +457,10 @@ class _SnapshotBuilder:
destination = self.destination / Path(*relative.parts)
os.symlink(target, destination)
after = os.stat(name, dir_fd=parent_descriptor, follow_symlinks=False)
if (
_mutation_identity(before) != _mutation_identity(after)
or os.readlink(
name,
dir_fd=parent_descriptor,
)
!= target
):
if _mutation_identity(before) != _mutation_identity(after) or os.readlink(
name,
dir_fd=parent_descriptor,
) != target:
raise SandboxError(f"dependency symlink changed while snapshotting: {relative}")
self.total_bytes += len(target_bytes)
self.budget.total_bytes += len(target_bytes)
@@ -524,15 +498,6 @@ class _SnapshotBuilder:
self.entries[entry.path] = entry
self.budget.entries += 1
def ensure_directory(self, relative: PurePosixPath) -> None:
"""Record and create one extra directory inside this snapshot.
Used for harness-owned mount points that must exist in the captured
bytes rather than be created against a read-only bind at runtime.
"""
self._record_directory(relative)
def finished_entries(self) -> tuple[AssetManifestEntry, ...]:
return tuple(sorted(self.entries.values(), key=lambda entry: entry.path.as_posix()))
@@ -597,7 +562,9 @@ def _sandbox_dependency_declarations(
or declaration.target_path in other.target_path.parents
or other.target_path in declaration.target_path.parents
):
raise SandboxError(f"sandbox dependency targets overlap: {declaration.target} and {other.target}")
raise SandboxError(
f"sandbox dependency targets overlap: {declaration.target} and {other.target}"
)
return tuple(declarations)
@@ -679,7 +646,9 @@ def _validate_dependency_symlinks(
)
if sandbox_resolved != sandbox_boundary and sandbox_boundary not in sandbox_resolved.parents:
raise SandboxError(f"dependency symlink escapes the sandbox workspace: {entry.path}")
manifest_resolved = PurePosixPath(posixpath.normpath((entry.path.parent / target).as_posix()))
manifest_resolved = PurePosixPath(
posixpath.normpath((entry.path.parent / target).as_posix())
)
if manifest_resolved != manifest_boundary and manifest_boundary not in manifest_resolved.parents:
continue
link = container / Path(*entry.path.parts)
@@ -1047,7 +1016,8 @@ def _dependency_mounts(
snapshot: TaskAssetSnapshot,
) -> list[ReadOnlyMount]:
declarations = tuple(
(declaration.source, declaration.target) for declaration in _sandbox_dependency_declarations(task)
(declaration.source, declaration.target)
for declaration in _sandbox_dependency_declarations(task)
)
if snapshot.dependency_declarations != declarations:
raise SandboxError("task asset snapshot does not match this dependency declaration")
@@ -1,7 +1,7 @@
{
"name": "gitnexus",
"description": "Code intelligence powered by a knowledge graph. Provides execution flow tracing, blast radius analysis, and augmented search across your codebase.",
"version": "1.6.10-rc.94",
"version": "1.6.9",
"author": {
"name": "GitNexus"
},
@@ -1,7 +1,7 @@
{
"name": "gitnexus",
"description": "Code intelligence powered by a knowledge graph. Provides execution flow tracing, blast radius analysis, and augmented search across your codebase.",
"version": "1.6.10-rc.94",
"version": "1.6.9",
"skills": "./skills",
"mcpServers": "./.mcp.json",
"hooks": "./hooks/hooks.json",
+53 -111
View File
@@ -11,14 +11,14 @@
"@langchain/anthropic": "^1.5.1",
"@langchain/core": "^1.2.2",
"@langchain/google-genai": "^2.2.0",
"@langchain/langgraph": "^1.4.8",
"@langchain/langgraph": "^1.4.7",
"@langchain/ollama": "^1.3.0",
"@langchain/openai": "^1.5.3",
"@sigma/edge-curve": "^3.1.0",
"@tailwindcss/vite": "^4.3.2",
"axios": "^1.18.1",
"d3": "^7.9.0",
"dompurify": "^3.4.12",
"dompurify": "^3.4.11",
"gitnexus-shared": "file:../gitnexus-shared",
"graphology": "^0.26.0",
"graphology-indices": "^0.17.0",
@@ -29,14 +29,14 @@
"i18next": "^26.3.0",
"i18next-browser-languagedetector": "^8.2.1",
"langchain": "^1.4.6",
"lru-cache": "^11.5.2",
"lru-cache": "^11.5.1",
"lucide-react": "^1.23.0",
"mermaid": "^11.15.0",
"mnemonist": "^0.40.4",
"pandemonium": "^2.4.0",
"react": "^19.2.5",
"react-dom": "^19.2.7",
"react-i18next": "^17.0.10",
"react-i18next": "^17.0.8",
"react-markdown": "^10.1.0",
"react-syntax-highlighter": "^16.1.1",
"react-zoom-pan-pinch": "^4.0.3",
@@ -47,7 +47,7 @@
"zod": "^4.4.3"
},
"devDependencies": {
"@babel/types": "^8.0.0",
"@babel/types": "^7.29.0",
"@playwright/test": "^1.61.1",
"@testing-library/jest-dom": "^6.9.1",
"@testing-library/react": "^16.3.2",
@@ -63,7 +63,7 @@
"jsdom": "^29.1.1",
"tree-sitter-wasms": "^0.1.13",
"typescript": "^5.4.5",
"vite": "^8.1.5",
"vite": "^8.1.4",
"vitest": "^4.1.10",
"wait-on": "^9.0.10"
},
@@ -186,13 +186,13 @@
}
},
"node_modules/@babel/helper-string-parser": {
"version": "8.0.0",
"resolved": "https://registry.npmjs.org/@babel/helper-string-parser/-/helper-string-parser-8.0.0.tgz",
"integrity": "sha512-6mJgmFFFIIO82vvoLt9XtRC7/TkzXfts1t/SpRX4IHSzMgqoPYCWesVu1udUPUWioAE/2fcG6WuI8zrkE1gwrg==",
"version": "7.29.7",
"resolved": "https://registry.npmjs.org/@babel/helper-string-parser/-/helper-string-parser-7.29.7.tgz",
"integrity": "sha512-Pb5ijPrZ89GDH8223L4UP8i6QApWxs04RbPQJTeWDV0/keR2E36MeKnyr6LYmUUvqRRI+Iv87SuF1W6ErINzYw==",
"dev": true,
"license": "MIT",
"engines": {
"node": "^22.18.0 || >=24.11.0"
"node": ">=6.9.0"
}
},
"node_modules/@babel/helper-validator-identifier": {
@@ -221,17 +221,16 @@
"node": ">=6.0.0"
}
},
"node_modules/@babel/parser/node_modules/@babel/helper-string-parser": {
"version": "7.29.7",
"resolved": "https://registry.npmjs.org/@babel/helper-string-parser/-/helper-string-parser-7.29.7.tgz",
"integrity": "sha512-Pb5ijPrZ89GDH8223L4UP8i6QApWxs04RbPQJTeWDV0/keR2E36MeKnyr6LYmUUvqRRI+Iv87SuF1W6ErINzYw==",
"dev": true,
"node_modules/@babel/runtime": {
"version": "7.29.2",
"resolved": "https://registry.npmjs.org/@babel/runtime/-/runtime-7.29.2.tgz",
"integrity": "sha512-JiDShH45zKHWyGe4ZNVRrCjBz8Nh9TMmZG1kh4QTK8hCBTWBi8Da+i7s1fJw7/lYpM4ccepSNfqzZ/QvABBi5g==",
"license": "MIT",
"engines": {
"node": ">=6.9.0"
}
},
"node_modules/@babel/parser/node_modules/@babel/types": {
"node_modules/@babel/types": {
"version": "7.29.7",
"resolved": "https://registry.npmjs.org/@babel/types/-/types-7.29.7.tgz",
"integrity": "sha512-4zBIxpPzowiZpusoFkyGVwakdRJUyuH5PxQ/PrqghfdFWWasvnCdPfQXHrenDai+gyLARulZjZowCOj6fjT4pA==",
@@ -245,39 +244,6 @@
"node": ">=6.9.0"
}
},
"node_modules/@babel/runtime": {
"version": "7.29.2",
"resolved": "https://registry.npmjs.org/@babel/runtime/-/runtime-7.29.2.tgz",
"integrity": "sha512-JiDShH45zKHWyGe4ZNVRrCjBz8Nh9TMmZG1kh4QTK8hCBTWBi8Da+i7s1fJw7/lYpM4ccepSNfqzZ/QvABBi5g==",
"license": "MIT",
"engines": {
"node": ">=6.9.0"
}
},
"node_modules/@babel/types": {
"version": "8.0.0",
"resolved": "https://registry.npmjs.org/@babel/types/-/types-8.0.0.tgz",
"integrity": "sha512-K8ponJDxBwDHigkeFqaqT5wLGl4bTlwMafR8k7b5CPxr6Ww+UG9ls8Yx6Tcpboxu97eeGVEEyKcHmEyOwN1vSw==",
"dev": true,
"license": "MIT",
"dependencies": {
"@babel/helper-string-parser": "^8.0.0",
"@babel/helper-validator-identifier": "^8.0.0"
},
"engines": {
"node": "^22.18.0 || >=24.11.0"
}
},
"node_modules/@babel/types/node_modules/@babel/helper-validator-identifier": {
"version": "8.0.4",
"resolved": "https://registry.npmjs.org/@babel/helper-validator-identifier/-/helper-validator-identifier-8.0.4.tgz",
"integrity": "sha512-4wFaiLd0bVo4cIoTXI3zKI038NIWE/cr3jvBjejOVYVxV/m8Ltav1USiGzG1fmS5J2RhgEOgXNNK46cRPnRsrg==",
"dev": true,
"license": "MIT",
"engines": {
"node": "^22.18.0 || >=24.11.0"
}
},
"node_modules/@bcoe/v8-coverage": {
"version": "1.0.2",
"resolved": "https://registry.npmjs.org/@bcoe/v8-coverage/-/v8-coverage-1.0.2.tgz",
@@ -1172,13 +1138,13 @@
}
},
"node_modules/@langchain/langgraph": {
"version": "1.4.8",
"resolved": "https://registry.npmjs.org/@langchain/langgraph/-/langgraph-1.4.8.tgz",
"integrity": "sha512-DN1Np1XefdBEbp1qBKlt39cwoL743AAGpR5Ipja0gY2YbWvsoQnOTIrjnj/orSAhaUYsdTKS8VSWdFzsHZo6Ig==",
"version": "1.4.7",
"resolved": "https://registry.npmjs.org/@langchain/langgraph/-/langgraph-1.4.7.tgz",
"integrity": "sha512-2tcyf3QGC7v89kqSxMCtRvzg/3L/4yHtOaWC49A8KieCciWJs7LGaxHoPB6QRxXyUgyR+Zg9Q1ss/XJIE+JuSQ==",
"license": "MIT",
"dependencies": {
"@langchain/langgraph-checkpoint": "^1.1.3",
"@langchain/langgraph-sdk": "~1.9.26",
"@langchain/langgraph-sdk": "~1.9.25",
"@langchain/protocol": "^0.0.18",
"@standard-schema/spec": "1.1.0"
},
@@ -1203,9 +1169,9 @@
}
},
"node_modules/@langchain/langgraph-sdk": {
"version": "1.9.28",
"resolved": "https://registry.npmjs.org/@langchain/langgraph-sdk/-/langgraph-sdk-1.9.28.tgz",
"integrity": "sha512-4j3XuM0PvtmAbL8mPfBS99ez3+ytRfgbOpAR/nOeaejTRF3Q9dNw2QnaGLGng8wLPtGLoSj+SYgUOVxy9Bv9vg==",
"version": "1.9.25",
"resolved": "https://registry.npmjs.org/@langchain/langgraph-sdk/-/langgraph-sdk-1.9.25.tgz",
"integrity": "sha512-mRKW8zyQUaHox+HirRFMRrPqOvNbQI3xeXDt6kkk4PbBg77V92bsO1WzUVNrmJ81zCkvxyOrWSK8D6ioCj0a8A==",
"license": "MIT",
"dependencies": {
"@langchain/protocol": "^0.0.18",
@@ -1242,9 +1208,9 @@
"license": "MIT"
},
"node_modules/@langchain/langgraph-sdk/node_modules/p-queue": {
"version": "9.3.3",
"resolved": "https://registry.npmjs.org/p-queue/-/p-queue-9.3.3.tgz",
"integrity": "sha512-NXAOdnEe5FsZJfT4oK84lE1Y5cFFdWlRuOo5tww8DyNMxyRXwn39fIkUtNLKppcPC+UYU/bXujNCUGDv01y7CA==",
"version": "9.3.0",
"resolved": "https://registry.npmjs.org/p-queue/-/p-queue-9.3.0.tgz",
"integrity": "sha512-7NED7xhQ74Ngp4JP/2e0VZHp7vSWfJfqeiR92jPgxsz6m0Se4P03YoTKa9dDXyZ3r6P616gUXttrB6nnHYKang==",
"license": "MIT",
"dependencies": {
"eventemitter3": "^5.0.4",
@@ -4109,9 +4075,9 @@
"peer": true
},
"node_modules/dompurify": {
"version": "3.4.12",
"resolved": "https://registry.npmjs.org/dompurify/-/dompurify-3.4.12.tgz",
"integrity": "sha512-zQvGet8Z2sWbQhCmfFz/T5QWH2oBmjnqK3qvOjaqaNLrLEF912WamU+ohnTp0TCep/MFVHpdJuCZEdFOdTnEFg==",
"version": "3.4.11",
"resolved": "https://registry.npmjs.org/dompurify/-/dompurify-3.4.11.tgz",
"integrity": "sha512-zhlUV12GsaRzMsf9q5M254YhA4+VuF0fG+QFqu6aYpoGlKtz+w8//jBcGVYBgQkR5GHjUomejY84AV+/uPbWdw==",
"license": "(MPL-2.0 OR Apache-2.0)",
"optionalDependencies": {
"@types/trusted-types": "^2.0.7"
@@ -4391,9 +4357,9 @@
"license": "Unlicense"
},
"node_modules/fast-uri": {
"version": "3.1.4",
"resolved": "https://registry.npmjs.org/fast-uri/-/fast-uri-3.1.4.tgz",
"integrity": "sha512-8JnbkQ4juDyvYs4mgFGQqg4yCYtFDtUtmp2QIQq11ZZe5CFQ5wcqm1rqDgAh/QdMySuBnPzMUiJUNZG5N/AiQw==",
"version": "3.1.2",
"resolved": "https://registry.npmjs.org/fast-uri/-/fast-uri-3.1.2.tgz",
"integrity": "sha512-rVjf7ArG3LTk+FS6Yw81V1DLuZl1bRbNrev6Tmd/9RaroeeRRJhAt7jg/6YFxbvAQXUCavSoZhPPj6oOx+5KjQ==",
"dev": true,
"funding": [
{
@@ -5706,9 +5672,9 @@
}
},
"node_modules/lru-cache": {
"version": "11.5.2",
"resolved": "https://registry.npmjs.org/lru-cache/-/lru-cache-11.5.2.tgz",
"integrity": "sha512-4pfM1Ff0x50o0tQwb5ucw/RzNyD0/YJME6IVcStalZuMWxdt3sR3huStTtxz4PUmvZfRguvDejasvQ2kifR11g==",
"version": "11.5.1",
"resolved": "https://registry.npmjs.org/lru-cache/-/lru-cache-11.5.1.tgz",
"integrity": "sha512-RPimw/7aMdv2oqRrxKwvZXcPfwBrn/JZ2xYcY9Hus/6LaS3VOAKVWKWgNLCFSiOm1ESXinjsDlidVU7JlnCN2A==",
"license": "BlueOak-1.0.0",
"engines": {
"node": "20 || >=22"
@@ -5755,30 +5721,6 @@
"source-map-js": "^1.2.1"
}
},
"node_modules/magicast/node_modules/@babel/helper-string-parser": {
"version": "7.29.7",
"resolved": "https://registry.npmjs.org/@babel/helper-string-parser/-/helper-string-parser-7.29.7.tgz",
"integrity": "sha512-Pb5ijPrZ89GDH8223L4UP8i6QApWxs04RbPQJTeWDV0/keR2E36MeKnyr6LYmUUvqRRI+Iv87SuF1W6ErINzYw==",
"dev": true,
"license": "MIT",
"engines": {
"node": ">=6.9.0"
}
},
"node_modules/magicast/node_modules/@babel/types": {
"version": "7.29.7",
"resolved": "https://registry.npmjs.org/@babel/types/-/types-7.29.7.tgz",
"integrity": "sha512-4zBIxpPzowiZpusoFkyGVwakdRJUyuH5PxQ/PrqghfdFWWasvnCdPfQXHrenDai+gyLARulZjZowCOj6fjT4pA==",
"dev": true,
"license": "MIT",
"dependencies": {
"@babel/helper-string-parser": "^7.29.7",
"@babel/helper-validator-identifier": "^7.29.7"
},
"engines": {
"node": ">=6.9.0"
}
},
"node_modules/make-dir": {
"version": "4.0.0",
"resolved": "https://registry.npmjs.org/make-dir/-/make-dir-4.0.0.tgz",
@@ -6884,9 +6826,9 @@
}
},
"node_modules/nanoid": {
"version": "3.3.16",
"resolved": "https://registry.npmjs.org/nanoid/-/nanoid-3.3.16.tgz",
"integrity": "sha512-bzlKTyNJ7+LdGIIwy8ijFpIqEQIvafahV7eYykJ8Cvh42EdJeODoJ6gUJXpQJvej1BddH8OqTXZNE/KfbWAu8Q==",
"version": "3.3.15",
"resolved": "https://registry.npmjs.org/nanoid/-/nanoid-3.3.15.tgz",
"integrity": "sha512-y7Wygv/7mEOvxTuEQDB8StXdMRBWf1kR/tlhAzBRUFkB2jfcLOAxO/SHmOO2zgz1pVgK29/kyupn059/bCHdjA==",
"funding": [
{
"type": "github",
@@ -7274,9 +7216,9 @@
}
},
"node_modules/postcss": {
"version": "8.5.22",
"resolved": "https://registry.npmjs.org/postcss/-/postcss-8.5.22.tgz",
"integrity": "sha512-KBDEIpLrvpv16pp3K0Fw+UCoZfopFjjgeB+0tA/aaThfEE74kKDLrgg603YvOWJyg3+WYtyq3xYsQWsIyZlPqQ==",
"version": "8.5.16",
"resolved": "https://registry.npmjs.org/postcss/-/postcss-8.5.16.tgz",
"integrity": "sha512-vuwillviilfKZsg0VGj5R/YwwcHx4SLsIOI/7K6mQkWx+l5cUHTjj5g0AasTBcyXsbfTgrwsUNmVUb5xVwyPwg==",
"funding": [
{
"type": "opencollective",
@@ -7293,7 +7235,7 @@
],
"license": "MIT",
"dependencies": {
"nanoid": "^3.3.16",
"nanoid": "^3.3.12",
"picocolors": "^1.1.1",
"source-map-js": "^1.2.1"
},
@@ -7414,9 +7356,9 @@
}
},
"node_modules/react-i18next": {
"version": "17.0.10",
"resolved": "https://registry.npmjs.org/react-i18next/-/react-i18next-17.0.10.tgz",
"integrity": "sha512-XneHftyYA774MJkkccSkZ5oKrUpCnXIPmxio3wemqrVzCRLWiGXOMbIzObrer03fNDEnm8g8R5yYls4HcE+esg==",
"version": "17.0.8",
"resolved": "https://registry.npmjs.org/react-i18next/-/react-i18next-17.0.8.tgz",
"integrity": "sha512-0ooKbGLU8JXhe1zwpQUWIeXSgLPOfwJmgheWRIUpcoA0CpyabpGhayjdG+/eA5esC1AQ8h2jWpXjJfzQzeDOCw==",
"license": "MIT",
"dependencies": {
"@babel/runtime": "^7.29.2",
@@ -7426,7 +7368,7 @@
"peerDependencies": {
"i18next": ">= 26.2.0",
"react": ">= 16.8.0",
"typescript": "^5 || ^6 || ^7"
"typescript": "^5 || ^6"
},
"peerDependenciesMeta": {
"react-dom": {
@@ -7952,9 +7894,9 @@
}
},
"node_modules/tar": {
"version": "7.5.20",
"resolved": "https://registry.npmjs.org/tar/-/tar-7.5.20.tgz",
"integrity": "sha512-9FcyK4PA6+WbzlTM9WhQm6vB5W7cP7dUiPsv1g7YDwEQnQ1CGpK3MGlKk/ITVWMk05kHZuBhmVhiv8LZoy/PFQ==",
"version": "7.5.16",
"resolved": "https://registry.npmjs.org/tar/-/tar-7.5.16.tgz",
"integrity": "sha512-56adEpPMouktRlBLXiaYFFzZ/3+JXa8P9n7WbR+ibIjtviN55mEaOkiysCnPnWm+7kkui1Dn8J9l+g6zV8731w==",
"dev": true,
"license": "BlueOak-1.0.0",
"dependencies": {
@@ -8354,15 +8296,15 @@
}
},
"node_modules/vite": {
"version": "8.1.5",
"resolved": "https://registry.npmjs.org/vite/-/vite-8.1.5.tgz",
"integrity": "sha512-7ULLwsCdYx/nRyrpiEwvqb5TFHrMVZyBt+rg/OAXT7rgj/z+DtTDyKFeLAdDkubDVDKD8jOsndmy7m55XcfUsw==",
"version": "8.1.4",
"resolved": "https://registry.npmjs.org/vite/-/vite-8.1.4.tgz",
"integrity": "sha512-bTT9PsdWO+MQMNG9ZXIP/qM9wGh37DFxTV/sPq9cFpHr3w4jkgef032PkAL9jAqhk3Nz8NQw3O8n6/xFkqO4QQ==",
"license": "MIT",
"dependencies": {
"lightningcss": "^1.32.0",
"picomatch": "^4.0.5",
"postcss": "^8.5.17",
"rolldown": "~1.1.5",
"postcss": "^8.5.16",
"rolldown": "~1.1.4",
"tinyglobby": "^0.2.17"
},
"bin": {
+6 -6
View File
@@ -21,14 +21,14 @@
"@langchain/anthropic": "^1.5.1",
"@langchain/core": "^1.2.2",
"@langchain/google-genai": "^2.2.0",
"@langchain/langgraph": "^1.4.8",
"@langchain/langgraph": "^1.4.7",
"@langchain/ollama": "^1.3.0",
"@langchain/openai": "^1.5.3",
"@sigma/edge-curve": "^3.1.0",
"@tailwindcss/vite": "^4.3.2",
"axios": "^1.18.1",
"d3": "^7.9.0",
"dompurify": "^3.4.12",
"dompurify": "^3.4.11",
"gitnexus-shared": "file:../gitnexus-shared",
"graphology": "^0.26.0",
"graphology-indices": "^0.17.0",
@@ -39,14 +39,14 @@
"i18next": "^26.3.0",
"i18next-browser-languagedetector": "^8.2.1",
"langchain": "^1.4.6",
"lru-cache": "^11.5.2",
"lru-cache": "^11.5.1",
"lucide-react": "^1.23.0",
"mermaid": "^11.15.0",
"mnemonist": "^0.40.4",
"pandemonium": "^2.4.0",
"react": "^19.2.5",
"react-dom": "^19.2.7",
"react-i18next": "^17.0.10",
"react-i18next": "^17.0.8",
"react-markdown": "^10.1.0",
"react-syntax-highlighter": "^16.1.1",
"react-zoom-pan-pinch": "^4.0.3",
@@ -57,7 +57,7 @@
"zod": "^4.4.3"
},
"devDependencies": {
"@babel/types": "^8.0.0",
"@babel/types": "^7.29.0",
"@playwright/test": "^1.61.1",
"@testing-library/jest-dom": "^6.9.1",
"@testing-library/react": "^16.3.2",
@@ -73,7 +73,7 @@
"jsdom": "^29.1.1",
"tree-sitter-wasms": "^0.1.13",
"typescript": "^5.4.5",
"vite": "^8.1.5",
"vite": "^8.1.4",
"vitest": "^4.1.10",
"wait-on": "^9.0.10"
},
+3 -24
View File
@@ -284,7 +284,6 @@ Set these env vars to use a remote OpenAI-compatible `/v1/embeddings` endpoint i
export GITNEXUS_EMBEDDING_URL=http://your-server:8080/v1
export GITNEXUS_EMBEDDING_MODEL=BAAI/bge-large-en-v1.5
export GITNEXUS_EMBEDDING_DIMS=1024 # optional, default 384
export GITNEXUS_EMBEDDING_REQUEST_DIMS=omit # optional: omit "dimensions", or an integer to override it
export GITNEXUS_EMBEDDING_API_KEY=your-key # optional, default: "unused"
export GITNEXUS_EMBEDDING_MAX_ATTEMPTS=3 # optional, total attempts (1-20)
export GITNEXUS_EMBEDDING_RETRY_CAP_MS=5000 # optional, maximum retry delay
@@ -292,15 +291,6 @@ export GITNEXUS_EMBEDDING_MIN_INTERVAL_MS=0 # optional, minimum request spacing
gitnexus analyze . --embeddings
```
`GITNEXUS_EMBEDDING_REQUEST_DIMS` controls only the `dimensions` field sent in
the request body, independently of `GITNEXUS_EMBEDDING_DIMS` (which still
validates the returned vector's length):
- `omit` (or `none`, `off`, `false`, `0`) — do not send `dimensions` at all, for
strict backends that return the right vector size but reject the field.
- a positive integer — send that value instead of `GITNEXUS_EMBEDDING_DIMS`.
- unset — send `GITNEXUS_EMBEDDING_DIMS` (the previous behavior).
Works with Infinity, vLLM, TEI, llama.cpp, Ollama, LM Studio, or OpenAI. Retry and pacing settings are provider-neutral; provider-specific limits should be supplied through configuration. When unset, local embeddings are used unchanged.
## Multi-Repo Support
@@ -484,7 +474,7 @@ Configure the behavior with these environment variables:
| `GITNEXUS_FTS_CJK_SEGMENTATION` | `none`, `bigram` | `none` | `bigram` inserts overlapping character-bigram boundaries into Chinese/Japanese Han-ideograph spans in `content`/`description` before FTS indexing, so LadybugDB's space-only tokenizer can see sub-phrase word boundaries. Scoped to CJK Unified Ideographs only — Japanese Hiragana/Katakana and Korean Hangul are not currently segmented. Unlike `GITNEXUS_FTS_STEMMER`, this rewrites stored text — enabling it on an already-indexed repo requires a full `gitnexus analyze --force`; neither `--repair-fts` nor a plain incremental `analyze` applies it to previously-indexed files. Set the same value wherever `analyze` and search-serving processes (CLI query, MCP server, web server) run. |
| `GITNEXUS_COMMUNITY_ENGINE` | `graphology`, `icebug`, `auto` | `graphology` | Community-detection engine used during analyze. `graphology` uses the bundled default path. `icebug` and `auto` currently behave identically: both try the experimental Icebug CSR path and fall back to Graphology if the optional native module is unavailable or incompatible. |
| `GITNEXUS_WAL_CHECKPOINT_THRESHOLD` | integer `>= -1` | `67108864` (64 MiB) | LadybugDB WAL auto-checkpoint threshold during analyze (bytes). Auto-checkpoint remains enabled; `-1` keeps Ladybug's stock ~16 MiB. Larger thresholds reduce checkpoint frequency but increase the WAL size at rotation time — choose a smaller value on disk-constrained environments. |
| `GITNEXUS_LBUG_BUFFER_POOL_SIZE` | integer `>= 0` (bytes) | min(2 GiB, 80% RAM) | LadybugDB buffer-pool ceiling for every GitNexus database (analyze, MCP server, serve, group bridges). Bounded so a long-lived `gitnexus mcp` process or a large incremental `analyze` cannot grow toward LadybugDB's native 80%-of-RAM default and OOM the host (#2557). `0` restores that native unbounded default; invalid values warn and fall back to the default. During `analyze` the pool is right-sized to the graph and, on non-4 KiB-page hosts (Apple Silicon 16 KiB, Ascend/aarch64 64 KiB), scaled by the page-size granule ratio up to min(2 GiB × pageSize/4 KiB, 80% RAM) (#2631); this env var overrides all of that as an absolute value. |
| `GITNEXUS_LBUG_BUFFER_POOL_SIZE` | integer `>= 0` (bytes) | min(2 GiB, 80% RAM) | LadybugDB buffer-pool ceiling for every GitNexus database (analyze, MCP server, serve, group bridges). Bounded so a long-lived `gitnexus mcp` process or a large incremental `analyze` cannot grow toward LadybugDB's native 80%-of-RAM default and OOM the host (#2557). `0` restores that native unbounded default; invalid values warn and fall back to the default. |
| `GITNEXUS_LBUG_MAX_DB_SIZE` | positive integer (bytes) | `17179869184` (16 GiB) | Upper bound for a single LadybugDB database file. This is an mmap/disk-address-space ceiling, not a memory limit — it does not constrain the buffer pool (use `GITNEXUS_LBUG_BUFFER_POOL_SIZE` for that). Raise it when indexing genuinely huge monorepos; invalid values silently fall back to the default. |
```bash
@@ -511,19 +501,9 @@ For very large repositories:
# Increase Node.js heap size
NODE_OPTIONS="--max-old-space-size=16384" npx gitnexus analyze
# Exclude large directories (this repo only)
# Exclude large directories
echo "vendor/" >> .gitnexusignore
echo "dist/" >> .gitnexusignore
# Exclude a directory across every repo you index, without touching each
# repo's own .gitnexusignore or needing push/commit access to it. GitNexus
# reads the same sources `git` itself does: core.excludesFile (all repos)
# and $GIT_DIR/info/exclude (this repo only, untracked). A repo's own
# .gitignore/.gitnexusignore can still override either with a `!pattern`
# negation. Skip both entirely with GITNEXUS_NO_GLOBAL_IGNORE=1.
git config --global core.excludesFile ~/.gitignore_global # applies to every repo
echo "docs/" >> ~/.gitignore_global
echo "build/" >> .git/info/exclude # this repo only, untracked
```
### Large files are being skipped
@@ -558,7 +538,7 @@ For repositories with very large source files, `GITNEXUS_WORKER_SUB_BATCH_MAX_BY
### Worker pool resilience tuning
Four env vars expose the pool's resilience layers (respawn budget, cumulative-timeout cap, circuit breaker, startup handshake). Defaults are tuned for typical repos; bump them when an analyze legitimately needs more retries, or lower them to fail-fast on a known-bad shape.
Three env vars expose the pool's resilience layers (respawn budget, cumulative-timeout cap, circuit breaker). Defaults are tuned for typical repos; bump them when an analyze legitimately needs more retries, or lower them to fail-fast on a known-bad shape.
| Variable | Default | Effect |
| ----------------------------------------------- | ----------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ |
@@ -566,7 +546,6 @@ Four env vars expose the pool's resilience layers (respawn budget, cumulative-ti
| `GITNEXUS_WORKER_MAX_CUMULATIVE_TIMEOUT_MS` | `5 × subBatchTimeoutMs` | Total retry wall-time budget per job before quarantining. Bounds exponentially-growing retry waits. |
| `GITNEXUS_WORKER_CONSECUTIVE_FAILURE_THRESHOLD` | `max(3, poolSize)` | Per-slot consecutive deaths before the pool's circuit breaker trips. After tripping, dispatches require a fresh pool. |
| `GITNEXUS_WORKER_SHUTDOWN_DRAIN_MS` | `30000` | Max wait at pool shutdown for a retired worker still inside native code — terminated at its next JS-safe point instead of mid-native-call, which would abort the process (`Napi::Error`, #2432). |
| `GITNEXUS_WORKER_READY_TIMEOUT_MS` | `5000` | Startup budget for a parse worker to load its grammar bindings and report `{type:'ready'}`. Slots that miss it are treated as startup crashes. Raise it on a slow or heavily loaded host where a full pool cold-starting concurrently needs more than 5s. |
| `GITNEXUS_CPP_CAPTURE_BUDGET_MS` | `20000` | Per-file wall-clock budget for C++ capture extraction; on breach the file keeps partial captures with a warning (#2432). `0` expires immediately. |
### Graph cleanup tuning
@@ -1 +1 @@
36e29abc0780bc857b6df6dd180a0b6036c8a28f927ccc2d4fe50eede24d0c99
a99e69ab2dfb897ed771c6a8e29c5b32843a7f734db701e0699afc07c090e4d5
+3 -6
View File
@@ -46,9 +46,8 @@
"_note": "#2046: F35 qualified-constructor captures now emit @reference.qualified-name + a simple-name @reference.name on `new Ns.Foo()`/`new A.B.Foo()`; namespace_declaration/file_scoped_namespace_declaration now emit @declaration.namespace name captures (feeding the non-destructive namespacePrefix sidecar for `new B.Foo()` same-tail disambiguation). + csharp-interface-only-base and csharp-namespace-qualified-ctor fixtures. Pure capture-additive + fixture-corpus drift; scaling stays linear (~1.11)."
},
"rust": {
"fingerprint": "f7742f65f14d7d6590df7f16303fc3cc9dc0c233cd80bf90c98b084933cd3846",
"fingerprint": "df369c5a5f8de7753fc8bab8b4108ef5081750974ea5085ba9a867675ac9eb29",
"scaling_budget": 1.5,
"_rebaselined_dyn_trait_object_2604": "#2604: RUST_SCOPE_QUERY now captures function_signature_item (abstract trait methods, no body) as a scope + declaration, so a &dyn Trait receiver can dispatch a CALLS edge to the trait's own method. Additive capture shift across every bench fixture with a required trait method. Prior df369c5a5f8de7753fc8bab8b4108ef5081750974ea5085ba9a867675ac9eb29 -> f7742f65f14d7d6590df7f16303fc3cc9dc0c233cd80bf90c98b084933cd3846; scaling 1.033 < 1.5.",
"_rebaselined_callable_flow_2522_review": "PR #2522 review hardening: callable operands retain expression/qualified identity and formals retain signature metadata. Prior 65e5bca66bb1ca117949409e8fb5c80ee69d6f1b5318908eaaecf08da0482e5c -> df369c5a5f8de7753fc8bab8b4108ef5081750974ea5085ba9a867675ac9eb29; scaling 1.065 < 1.5.",
"_rebaselined_callable_flow_2522_followup": "PR #2522 follow-up: Rust fn-value callable flow facts with invocation/constructor-result suppression. Prior ac610bbe97666bf285923479dd7b43a2fe4c5354aae8df1bcbafdc04fb220f82 -> 65e5bca66bb1ca117949409e8fb5c80ee69d6f1b5318908eaaecf08da0482e5c; scaling 1.024 < 1.5.",
"_rebaselined": "#1956 tri-review U1: rust-qualified-trait fixture (scoped + generic-of-scoped impl trait paths); bareTypeIdentifier now resolves scoped_type_identifier bases by their name: tail (additive, no existing-fixture drift); linear (~1.04). #1975: + rust-scoped-impl fixture (impl a::Inner / b::Inner inherent scoped impls) \u2014 legacy @definition.impl scoped arm + findEnclosingClassInfo inherent-impl scoped target; rust scope-extractor captures byte-identical. | #942: scope-resolution-only cleanup reworded fixture comments; capture byte-positions shift, capture LOGIC unchanged.",
@@ -90,7 +89,7 @@
"_rebaselined": "#1919 review CF3 fix: extended kotlin-local-property-owner (init/accessor destructuring) + new dart-accessor-owner fixture (getter/setter ownership). Fingerprint-only corpus drift; scaling ~1.0."
},
"java": {
"fingerprint": "d04298a91beec76d0fa7099b3d71265723be60c1df688969aa954f135dd49686",
"fingerprint": "975b68aaac6d06094260fb0c67f9b1bc03692ba7220669d192aca9dccd5fc0ca",
"scaling_budget": 1.5,
"_rebaselined_callable_flow_2522_review": "PR #2522 review hardening: callable operands retain expression/qualified identity and formals retain signature metadata; same-name lexical regions use an O(ancestor-depth) ID-set lookup. Prior d5c59d7dc9e206637515d5aea1163f7c1cdd76410c38c5fe6143d13d19677d6a -> 004a3592998dca1193bd1429a8284513725de7764f2a3eceedaaa984cfd763b4; scaling 0.992 < 1.5.",
"_rebaselined_callable_flow_2522_followup": "PR #2522 follow-up: Java method-reference/SAM callable flow facts with invocation-result suppression. Prior 062d754764aaa8a6772fb90875c710502a63e3e7a300e633942381ed914faada -> d5c59d7dc9e206637515d5aea1163f7c1cdd76410c38c5fe6143d13d19677d6a; scaling 1.074 < 1.5.",
@@ -98,9 +97,7 @@
"_note": "#1928 / #2045: F35 adds qualified + qualified-generic constructor query captures (`new pkg.Foo()`, `new a.b.Foo()`, `new pkg.Box<T>()`); F38 synthesizes `@reference.call.constructor` on `super(...)`/`this(...)` explicit_constructor_invocation nodes; F41 generic-aware stripQualifier in interpret (type-binding normalization). + java-qualified-constructor and java-explicit-constructor fixtures. Pure capture-additive + fixture-corpus drift; scaling stays linear (~1.06).",
"_rebaselined_2522_review_fixes": "PR #2522 review fixes: get/test dropped from callableProtocolMethods. Prior 004a3592998dca1193bd1429a8284513725de7764f2a3eceedaaa984cfd763b4 -> f3b4f4b6610e07c3ac90deb1c53d3572b6ad55a36e5d7134984876d30031ff67; scaling ratio re-verified within budget.",
"_rebaselined_2550_instance_model": "PR #2549 (#2550): anonymous class bodies emit synthesized @declaration.class/@declaration.name (Worker$N), an @reference.inherits to the constructed type, and receiver @type-binding.* captures; six new java-* fixtures joined the corpus. Prior f3b4f4b6610e07c3ac90deb1c53d3572b6ad55a36e5d7134984876d30031ff67 -> d79c3b92acfc866094981499b977388ca14f90839bca0c040342ab1cec00aa90; scaling 1.058 < 1.5.",
"_rebaselined_2555_enum_constant_bodies": "PR for #2555: enum constant bodies emit synthesized E$N classes + @reference.inherits to the host enum; anonymous naming follows JLS 13.1 immediately-enclosing-type chains INCLUDING anonymous enclosing types (NestHost$1$1, N$1$1); six new java-* fixtures joined the corpus. Prior d79c3b92acfc866094981499b977388ca14f90839bca0c040342ab1cec00aa90 -> 975b68aaac6d06094260fb0c67f9b1bc03692ba7220669d192aca9dccd5fc0ca; scaling 1.05 < 1.5.",
"_rebaselined_2564_record_capture": "PR for #2564: JAVA_QUERIES gained a (record_declaration name: (identifier) @name) @definition.record capture, previously entirely missing (record_declaration had no structure-phase capture at all, unlike class/interface/enum) - a record's methods existed as ownerless Method nodes with no HAS_METHOD edge. Two new java-* fixtures (java-record-methods, java-new-expr-chain-call) joined the corpus. Prior 975b68aaac6d06094260fb0c67f9b1bc03692ba7220669d192aca9dccd5fc0ca -> 85fc7af9c3c1bceac76cb4f27214410b04967682a2eaa7e468e26efd1f4e2537; scaling 1.059 < 1.5.",
"_rebaselined_2561_enum_constant_receiver": "PR for #2561: synthesizeJavaAnonymousClassDeclarations now emits a class-scope @type-binding.annotation/name/type per enum constant (constant simple name -> its E$N synthesized class when bodied, else the host enum) so E.CONST.method() resolves through the existing compound-receiver chain walk. Two drivers of the drift, both in the java-enum-constant-body fixture (this bench's corpus IS test/fixtures/lang-resolution): (1) one extra type-binding match per enum_constant from the capture change; (2) review follow-up added a body-less Plain.java enum + EnumConst.dispatchToConstant/dispatchInherited methods (bodied-override, inherited-via-MRO, and body-less dispatch call sites). The review's fail-safe hardening (bodied constant binds ONLY to E$N, never the host enum, when name synthesis fails on a malformed tree) is output-neutral on this well-formed corpus (verified: fingerprint identical with and without it). Prior 85fc7af9c3c1bceac76cb4f27214410b04967682a2eaa7e468e26efd1f4e2537 -> d04298a91beec76d0fa7099b3d71265723be60c1df688969aa954f135dd49686; scaling < 1.5."
"_rebaselined_2555_enum_constant_bodies": "PR for #2555: enum constant bodies emit synthesized E$N classes + @reference.inherits to the host enum; anonymous naming follows JLS 13.1 immediately-enclosing-type chains INCLUDING anonymous enclosing types (NestHost$1$1, N$1$1); six new java-* fixtures joined the corpus. Prior d79c3b92acfc866094981499b977388ca14f90839bca0c040342ab1cec00aa90 -> 975b68aaac6d06094260fb0c67f9b1bc03692ba7220669d192aca9dccd5fc0ca; scaling 1.05 < 1.5."
},
"typescript": {
"fingerprint": "3280b13d3f9378ab23eee31c2edc779b5a9ae1e7bb510c23a24855b44406d2f4",
+81 -74
View File
@@ -1,16 +1,16 @@
{
"name": "gitnexus",
"version": "1.6.10-rc.94",
"version": "1.6.9",
"lockfileVersion": 3,
"requires": true,
"packages": {
"": {
"name": "gitnexus",
"version": "1.6.10-rc.94",
"version": "1.6.9",
"hasInstallScript": true,
"license": "PolyForm-Noncommercial-1.0.0",
"dependencies": {
"@ladybugdb/core": "^0.18.3",
"@ladybugdb/core": "^0.18.0",
"@modelcontextprotocol/sdk": "^1.0.0",
"@scarf/scarf": "^1.4.0",
"busboy": "^1.6.0",
@@ -24,7 +24,7 @@
"graphology-indices": "^0.17.0",
"graphology-utils": "^2.3.0",
"ignore": "^7.0.5",
"js-yaml": "^5.0.0",
"js-yaml": "^4.1.1",
"jsonc-parser": "^3.3.1",
"mnemonist": "^0.40.3",
"node-addon-api": "^8.0.0",
@@ -58,7 +58,9 @@
"@types/cli-progress": "^3.11.6",
"@types/cors": "^2.8.17",
"@types/express": "^5.0.6",
"@types/node": "^26.0.0",
"@types/js-yaml": "^4.0.9",
"@types/node": "^25.6.0",
"@types/uuid": "^11.0.0",
"@vitest/coverage-v8": "^4.0.18",
"gitnexus-shared": "file:../gitnexus-shared",
"tsx": "^4.0.0",
@@ -66,7 +68,7 @@
"vitest": "^4.0.18"
},
"engines": {
"node": "^22.18.0 || >=24.11.0"
"node": ">=22.0.0"
},
"optionalDependencies": {
"@huggingface/transformers": "^4.1.0",
@@ -1252,9 +1254,9 @@
}
},
"node_modules/@ladybugdb/core": {
"version": "0.18.3",
"resolved": "https://registry.npmjs.org/@ladybugdb/core/-/core-0.18.3.tgz",
"integrity": "sha512-XjpPKW4MrL28D2gYGTZuIjiEcPx12L21lx58QggrdrItw8o/e9Lmg/Ejoo4Kz08lZj+rIcC1Fu9thzIYOTUlJw==",
"version": "0.18.1",
"resolved": "https://registry.npmjs.org/@ladybugdb/core/-/core-0.18.1.tgz",
"integrity": "sha512-0c1kXDpdv7z/GB0oyFYnLEjLsXFwPHz1YD4wxtrk9hav8zJX5T1PHQMr+XRfdDI1NQjx4iNdbPQGGT7Bx/X2aw==",
"hasInstallScript": true,
"license": "MIT",
"dependencies": {
@@ -1263,17 +1265,17 @@
"node-addon-api": "^6.0.0"
},
"optionalDependencies": {
"@ladybugdb/core-darwin-arm64": "0.18.3",
"@ladybugdb/core-darwin-x64": "0.18.3",
"@ladybugdb/core-linux-arm64": "0.18.3",
"@ladybugdb/core-linux-x64": "0.18.3",
"@ladybugdb/core-win32-x64": "0.18.3"
"@ladybugdb/core-darwin-arm64": "0.18.1",
"@ladybugdb/core-darwin-x64": "0.18.1",
"@ladybugdb/core-linux-arm64": "0.18.1",
"@ladybugdb/core-linux-x64": "0.18.1",
"@ladybugdb/core-win32-x64": "0.18.1"
}
},
"node_modules/@ladybugdb/core-darwin-arm64": {
"version": "0.18.3",
"resolved": "https://registry.npmjs.org/@ladybugdb/core-darwin-arm64/-/core-darwin-arm64-0.18.3.tgz",
"integrity": "sha512-DGZTOlvSS4esEb1vTekY5IDoAvZAeYzR5cXVkECtQj9BVkk05zsvCAdTPo1Rz1BuI0qvqUVF+2WlIerI67iA2g==",
"version": "0.18.1",
"resolved": "https://registry.npmjs.org/@ladybugdb/core-darwin-arm64/-/core-darwin-arm64-0.18.1.tgz",
"integrity": "sha512-M5YZuAONRAv3awkr+cfaibn9Da+3pgDzRiek/JabWQuz48xgzW3Vh9yQH4s8Dq/bfQo6YTsaLIBRcUCCUzCtcg==",
"cpu": [
"arm64"
],
@@ -1284,9 +1286,9 @@
]
},
"node_modules/@ladybugdb/core-darwin-x64": {
"version": "0.18.3",
"resolved": "https://registry.npmjs.org/@ladybugdb/core-darwin-x64/-/core-darwin-x64-0.18.3.tgz",
"integrity": "sha512-Qp6j0CM/orBlK6KD0p/s4ofkIhNUwi1hdCgMw+fj81UHugWHkVLiYV4grRBdHhyplw+snchZpTxvfpxFbkG1Cw==",
"version": "0.18.1",
"resolved": "https://registry.npmjs.org/@ladybugdb/core-darwin-x64/-/core-darwin-x64-0.18.1.tgz",
"integrity": "sha512-kq+pyTskfCx++Mrbk7QssE/f/CpSuU50T8lhRtv4PaOKhC2Jf8/wAUOA17UxI594wAru3ERpqVBFUBWGcPk2ag==",
"cpu": [
"x64"
],
@@ -1297,9 +1299,9 @@
]
},
"node_modules/@ladybugdb/core-linux-arm64": {
"version": "0.18.3",
"resolved": "https://registry.npmjs.org/@ladybugdb/core-linux-arm64/-/core-linux-arm64-0.18.3.tgz",
"integrity": "sha512-F9miYjBuS43I7uNG199FNMqwdHJ98WA6dU3v2SZCeLXmXCdRzmYcuHQWlbNr2Tba9CX58w2XvBZoUaXZKJ/yKQ==",
"version": "0.18.1",
"resolved": "https://registry.npmjs.org/@ladybugdb/core-linux-arm64/-/core-linux-arm64-0.18.1.tgz",
"integrity": "sha512-fu7ke1haa5rPINcQn0+kxQijZ0A8ZDWP9e+X8xcDH94RagDbPWwG8yFC890cGSdc/j7mTV+xkA/y/kVHpmVI6w==",
"cpu": [
"arm64"
],
@@ -1310,9 +1312,9 @@
]
},
"node_modules/@ladybugdb/core-linux-x64": {
"version": "0.18.3",
"resolved": "https://registry.npmjs.org/@ladybugdb/core-linux-x64/-/core-linux-x64-0.18.3.tgz",
"integrity": "sha512-AfG5RDp/f/IDctDMpTAT5+2MYNtlWT191xiQNjSaWD4X85DhY3Dzps8Qu5VteIAPih5d6mmoaKGs8q0XIjfkFA==",
"version": "0.18.1",
"resolved": "https://registry.npmjs.org/@ladybugdb/core-linux-x64/-/core-linux-x64-0.18.1.tgz",
"integrity": "sha512-qp5HilHzDGuArfOyD+VyA7lVJ7IwQDKd81NZKKTmUwIAOJtdwqniYx6JZICPnlr36zFJBx/lGYoSsEzbC+TVdw==",
"cpu": [
"x64"
],
@@ -1323,9 +1325,9 @@
]
},
"node_modules/@ladybugdb/core-win32-x64": {
"version": "0.18.3",
"resolved": "https://registry.npmjs.org/@ladybugdb/core-win32-x64/-/core-win32-x64-0.18.3.tgz",
"integrity": "sha512-bHuFk0m9cnq0WGd9I4D8or8g6cC/BS58iatMtilqM3JpDPIQIFk6MQl6exL7P4xyWbkLwQgsrv2ToDnyoQNKvg==",
"version": "0.18.1",
"resolved": "https://registry.npmjs.org/@ladybugdb/core-win32-x64/-/core-win32-x64-0.18.1.tgz",
"integrity": "sha512-vHcXr7Df2X1dbb5ORK+SBmNstd/3tApGFImbAnaWiTuLDFlAdfY8lbiSBSp3OgFjc0BB7F3GYUUdvgDRJjK3zA==",
"cpu": [
"x64"
],
@@ -1929,6 +1931,13 @@
"dev": true,
"license": "MIT"
},
"node_modules/@types/js-yaml": {
"version": "4.0.9",
"resolved": "https://registry.npmjs.org/@types/js-yaml/-/js-yaml-4.0.9.tgz",
"integrity": "sha512-k4MGaQl5TGo/iipqb2UDG2UwjXziSWkh0uysQelTlJpX1qGlpUZYm8PnO4DxG1qBomtJUdYJ6qR6xdIah10JLg==",
"dev": true,
"license": "MIT"
},
"node_modules/@types/jsesc": {
"version": "2.5.1",
"resolved": "https://registry.npmjs.org/@types/jsesc/-/jsesc-2.5.1.tgz",
@@ -1937,13 +1946,13 @@
"license": "MIT"
},
"node_modules/@types/node": {
"version": "26.1.1",
"resolved": "https://registry.npmjs.org/@types/node/-/node-26.1.1.tgz",
"integrity": "sha512-nxAkRSVkN1Y0JC1W8ky/fTfkGsMmcrRsbx+3XoZE+rMOX71kLYTV7fLXpqud1GpbpP5TuffXFqfX7fH2GgZREw==",
"version": "25.9.5",
"resolved": "https://registry.npmjs.org/@types/node/-/node-25.9.5.tgz",
"integrity": "sha512-OScDchr2fwuUmWdf4kZ9h7PcJiYDVInhJizG/biAq3cAvqwYktuy/TYGGdZNMtNTFUP7rnb0NU4TUdm82kt4Rg==",
"devOptional": true,
"license": "MIT",
"dependencies": {
"undici-types": "~8.3.0"
"undici-types": ">=7.24.0 <7.24.7"
}
},
"node_modules/@types/qs": {
@@ -1981,6 +1990,17 @@
"@types/node": "*"
}
},
"node_modules/@types/uuid": {
"version": "11.0.0",
"resolved": "https://registry.npmjs.org/@types/uuid/-/uuid-11.0.0.tgz",
"integrity": "sha512-HVyk8nj2m+jcFRNazzqyVKiZezyhDKrGUA3jlEcg/nZ6Ms+qHwocba1Y/AaVaznJTAM9xpdFSh+ptbNrhOGvZA==",
"deprecated": "This is a stub types definition. uuid provides its own type definitions, so you do not need this installed.",
"dev": true,
"license": "MIT",
"dependencies": {
"uuid": "*"
}
},
"node_modules/@vitest/coverage-v8": {
"version": "4.1.10",
"resolved": "https://registry.npmjs.org/@vitest/coverage-v8/-/coverage-v8-4.1.10.tgz",
@@ -2296,20 +2316,20 @@
}
},
"node_modules/body-parser": {
"version": "2.3.0",
"resolved": "https://registry.npmjs.org/body-parser/-/body-parser-2.3.0.tgz",
"integrity": "sha512-2cGmJupaNgg+QUwVLAucDuWuoMZ6EX9iHDRswZ5lsNYEmwPaRknMPCLZz07yTzVq/83p4o/wzbDZbBrTvGGTIw==",
"version": "2.2.2",
"resolved": "https://registry.npmjs.org/body-parser/-/body-parser-2.2.2.tgz",
"integrity": "sha512-oP5VkATKlNwcgvxi0vM0p/D3n2C3EReYVX+DNYs5TjZFn/oQt2j+4sVJtSMr18pdRr8wjTcBl6LoV+FUwzPmNA==",
"license": "MIT",
"dependencies": {
"bytes": "^3.1.2",
"content-type": "^2.0.0",
"content-type": "^1.0.5",
"debug": "^4.4.3",
"http-errors": "^2.0.1",
"iconv-lite": "^0.7.2",
"http-errors": "^2.0.0",
"iconv-lite": "^0.7.0",
"on-finished": "^2.4.1",
"qs": "^6.15.2",
"raw-body": "^3.0.2",
"type-is": "^2.1.0"
"qs": "^6.14.1",
"raw-body": "^3.0.1",
"type-is": "^2.0.1"
},
"engines": {
"node": ">=18"
@@ -2319,23 +2339,10 @@
"url": "https://opencollective.com/express"
}
},
"node_modules/body-parser/node_modules/content-type": {
"version": "2.0.0",
"resolved": "https://registry.npmjs.org/content-type/-/content-type-2.0.0.tgz",
"integrity": "sha512-j/O/d7GcZCyNl7/hwZAb606rzqkyvaDctLmckbxLzHvFBzTJHuGEdodATcP3yIRoDrLHkIATJuvzbFlp/ki2cQ==",
"license": "MIT",
"engines": {
"node": ">=18"
},
"funding": {
"type": "opencollective",
"url": "https://opencollective.com/express"
}
},
"node_modules/brace-expansion": {
"version": "5.0.7",
"resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-5.0.7.tgz",
"integrity": "sha512-7oFy703dxfY3/NLxC1fh2SUCQ0H9rmAY+5EpDVfXjUTTs+HEwR2nYaqLv+GWcTsumwxPfiz6CzCNkwXwBUwqCA==",
"version": "5.0.6",
"resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-5.0.6.tgz",
"integrity": "sha512-kLpxurY4Z4r9sgMsyG0Z9uzsBlgiU/EFKhj/h91/8yHu0edo7XuixOIH3VcJ8kkxs6/jPzoI6U9Vj3WqbMQ94g==",
"license": "MIT",
"dependencies": {
"balanced-match": "^4.0.2"
@@ -3042,9 +3049,9 @@
"license": "MIT"
},
"node_modules/fast-uri": {
"version": "3.1.4",
"resolved": "https://registry.npmjs.org/fast-uri/-/fast-uri-3.1.4.tgz",
"integrity": "sha512-8JnbkQ4juDyvYs4mgFGQqg4yCYtFDtUtmp2QIQq11ZZe5CFQ5wcqm1rqDgAh/QdMySuBnPzMUiJUNZG5N/AiQw==",
"version": "3.1.2",
"resolved": "https://registry.npmjs.org/fast-uri/-/fast-uri-3.1.2.tgz",
"integrity": "sha512-rVjf7ArG3LTk+FS6Yw81V1DLuZl1bRbNrev6Tmd/9RaroeeRRJhAt7jg/6YFxbvAQXUCavSoZhPPj6oOx+5KjQ==",
"funding": [
{
"type": "github",
@@ -3403,9 +3410,9 @@
"license": "MIT"
},
"node_modules/hono": {
"version": "4.12.31",
"resolved": "https://registry.npmjs.org/hono/-/hono-4.12.31.tgz",
"integrity": "sha512-zJIHFrl6bq3RDd2YusFNCDlM8qUprxKswyi/OPzPyzKDdyBXDqWx8bZlZ7R+saTdSTatUmb3O7K4SspGPaEOQg==",
"version": "4.12.26",
"resolved": "https://registry.npmjs.org/hono/-/hono-4.12.26.tgz",
"integrity": "sha512-uyZtpnYxM9CmQ7QsQknM4zN8EftNqhON1qYeIKM0Se67CCEe2c44xyGURwB0axX2fBDu1dqHrHAc1hmNT8ITkw==",
"license": "MIT",
"engines": {
"node": ">=16.9.0"
@@ -3582,9 +3589,9 @@
"license": "MIT"
},
"node_modules/js-yaml": {
"version": "5.0.0",
"resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-5.0.0.tgz",
"integrity": "sha512-GSvaPUbk1U+FMZ7rJzF+F8e5YVtu7KnD40et/5rBXXRBv2jCO9L3qCewvIDDdudC0QycTFlf6EAA+h3kxBsuUw==",
"version": "4.3.0",
"resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-4.3.0.tgz",
"integrity": "sha512-1td788aAnnZ5qs7V2QIRl1owjtYpbKt749Y3xauqQgwIIGF/xXWz1wMTEBx5O3LK3lXLVuqXPdPxj2BoFHaW9Q==",
"funding": [
{
"type": "github",
@@ -3600,7 +3607,7 @@
"argparse": "^2.0.1"
},
"bin": {
"js-yaml": "bin/js-yaml.mjs"
"js-yaml": "bin/js-yaml.js"
}
},
"node_modules/jsesc": {
@@ -5128,9 +5135,9 @@
}
},
"node_modules/tar": {
"version": "7.5.20",
"resolved": "https://registry.npmjs.org/tar/-/tar-7.5.20.tgz",
"integrity": "sha512-9FcyK4PA6+WbzlTM9WhQm6vB5W7cP7dUiPsv1g7YDwEQnQ1CGpK3MGlKk/ITVWMk05kHZuBhmVhiv8LZoy/PFQ==",
"version": "7.5.16",
"resolved": "https://registry.npmjs.org/tar/-/tar-7.5.16.tgz",
"integrity": "sha512-56adEpPMouktRlBLXiaYFFzZ/3+JXa8P9n7WbR+ibIjtviN55mEaOkiysCnPnWm+7kkui1Dn8J9l+g6zV8731w==",
"license": "BlueOak-1.0.0",
"dependencies": {
"@isaacs/fs-minipass": "^4.0.0",
@@ -5503,9 +5510,9 @@
}
},
"node_modules/undici-types": {
"version": "8.3.0",
"resolved": "https://registry.npmjs.org/undici-types/-/undici-types-8.3.0.tgz",
"integrity": "sha512-j375ScV60dom+YkPFIfTLcOiPxkN/buHz5GobjLhixFuANaNs3C9l4GmrWqejgXWJ7BbJcFYpTEUkS1Ge8bpZQ==",
"version": "7.24.6",
"resolved": "https://registry.npmjs.org/undici-types/-/undici-types-7.24.6.tgz",
"integrity": "sha512-WRNW+sJgj5OBN4/0JpHFqtqzhpbnV0GuB+OozA9gCL7a993SmU+1JBZCzLNxYsbMfIeDL+lTsphD5jN5N+n0zg==",
"devOptional": true,
"license": "MIT"
},
+7 -5
View File
@@ -1,6 +1,6 @@
{
"name": "gitnexus",
"version": "1.6.10-rc.94",
"version": "1.6.9",
"description": "Graph-powered code intelligence for AI agents. Index any codebase, query via MCP or CLI.",
"author": "Abhigyan Patwari",
"license": "PolyForm-Noncommercial-1.0.0",
@@ -56,7 +56,7 @@
"version": "node scripts/sync-plugin-manifests.mjs"
},
"dependencies": {
"@ladybugdb/core": "^0.18.3",
"@ladybugdb/core": "^0.18.0",
"@modelcontextprotocol/sdk": "^1.0.0",
"@scarf/scarf": "^1.4.0",
"busboy": "^1.6.0",
@@ -70,7 +70,7 @@
"graphology-indices": "^0.17.0",
"graphology-utils": "^2.3.0",
"ignore": "^7.0.5",
"js-yaml": "^5.0.0",
"js-yaml": "^4.1.1",
"jsonc-parser": "^3.3.1",
"mnemonist": "^0.40.3",
"node-addon-api": "^8.0.0",
@@ -105,7 +105,9 @@
"@types/cli-progress": "^3.11.6",
"@types/cors": "^2.8.17",
"@types/express": "^5.0.6",
"@types/node": "^26.0.0",
"@types/js-yaml": "^4.0.9",
"@types/node": "^25.6.0",
"@types/uuid": "^11.0.0",
"@vitest/coverage-v8": "^4.0.18",
"gitnexus-shared": "file:../gitnexus-shared",
"tsx": "^4.0.0",
@@ -118,6 +120,6 @@
}
},
"engines": {
"node": "^22.18.0 || >=24.11.0"
"node": ">=22.0.0"
}
}
-6
View File
@@ -117,12 +117,6 @@ const LBUG_NATIVE = [
// to a live native DB, rm-then-rename over an existing parked copy) before
// any open — rename semantics are exactly what differs on Windows.
'test/unit/incremental-dirty-recovery.test.ts',
// #2623: the incremental writeback must load VECTOR before the CodeEmbedding
// join-delete, and the blocked path must escalate instead of crashing. The
// win32 VECTOR gate was removed in the same PR, so this ordering must be
// proven on the windows-latest native addon, not just Ubuntu. Budget: ~25s
// on Linux → expect ~2min on the slowest Windows shard.
'test/unit/incremental-vector-extension-ordering.test.ts',
];
// Process spawning and CLI tests — exercise child_process with real
+3 -14
View File
@@ -1,6 +1,6 @@
/**
* Install the LadybugDB FTS and VECTOR extensions into the shared home (~/.lbdb)
* up front, so every test in a sharded CI run finds them regardless of shard.
* Install the LadybugDB FTS extension into the shared home (~/.lbdb) up front, so
* every test in a sharded CI run finds it regardless of which shard it lands in.
*
* FTS-dependent tests split two ways: the LOAD-path gate (skipUnlessFtsAvailable)
* self-installs on miss, but the FILE-path gate (requireFtsResourceOrSkip, e.g.
@@ -17,24 +17,13 @@
import { mkdtempSync, rmSync } from 'node:fs';
import { tmpdir } from 'node:os';
import { join } from 'node:path';
import {
initLbug,
loadFTSExtension,
loadVectorExtension,
closeLbug,
} from '../src/core/lbug/lbug-adapter.js';
import { initLbug, loadFTSExtension, closeLbug } from '../src/core/lbug/lbug-adapter.js';
const dir = mkdtempSync(join(tmpdir(), 'gn-ensure-fts-'));
try {
await initLbug(join(dir, 'ensure-fts.lbug'));
const ok = await loadFTSExtension(undefined, { policy: 'auto' });
console.log(ok ? 'FTS extension ready.' : 'FTS extension unavailable (continuing).');
// VECTOR rides the same pre-install (#2623): the win32 gate is gone, so the
// vector suites genuinely run on Windows/macOS — installing once here means
// every sharded test process LOADs from ~/.lbdb instead of racing its own
// out-of-process INSTALL (bounded 15s each when the server is unreachable).
const vec = await loadVectorExtension(undefined, { policy: 'auto' });
console.log(vec ? 'VECTOR extension ready.' : 'VECTOR extension unavailable (continuing).');
} catch (err) {
console.warn(`ensure-fts: skipped (${err instanceof Error ? err.message : String(err)})`);
} finally {
-9
View File
@@ -18,7 +18,6 @@ import { boundedCheckpointBeforeExit } from '../core/lbug/shutdown-helpers.js';
import {
getOsPageSize,
isLbugCheckpointIoError,
isLbugCheckpointBusyError,
isLbugPageSizeFrameError,
isPageSizeAwareLadybug,
isWalCorruptionError,
@@ -1625,16 +1624,8 @@ const analyzeCommandImpl = async (
}
if (isLbugCheckpointIoError(err)) {
// #2599: when the checkpoint IO error also looks busy/locked, another
// handle holds the store open — name that actionable cause alongside the
// threshold hint (the original error is preserved so the hint still fires).
const heldOpen = isLbugCheckpointBusyError(err)
? ` Another process may hold the store open (a running \`gitnexus mcp\` server, or a\n` +
` stale reader) — close other GitNexus processes on this repo, then retry.\n`
: '';
cliError(
` LadybugDB failed while rotating/removing WAL checkpoint files.\n` +
heldOpen +
` This can happen when auto-checkpoint runs at the default threshold (~16MB).\n` +
` Retry with a larger checkpoint threshold to reduce checkpoint frequency:\n` +
` gitnexus analyze --wal-checkpoint-threshold ${RECOMMENDED_WAL_CHECKPOINT_THRESHOLD}\n` +
+4 -57
View File
@@ -12,16 +12,8 @@ import {
type EmbeddingRuntimeResolution,
} from '../core/embeddings/runtime-install.js';
import { cudaRedirectDoctorStatus } from '../core/embeddings/onnxruntime-node-resolver.js';
import {
checkLbugNative,
probeFtsExtensionLoad,
probeVectorExtensionLoad,
} from '../core/lbug/native-check.js';
import {
getEffectiveBufferPoolSize,
getOsPageSize,
isPageSizeAwareLadybug,
} from '../core/lbug/lbug-config.js';
import { checkLbugNative, probeFtsExtensionLoad } from '../core/lbug/native-check.js';
import { getOsPageSize, isPageSizeAwareLadybug } from '../core/lbug/lbug-config.js';
import { diagnoseExtensionLoad } from '../core/lbug/extension-load-error.js';
import { getExtensionInstallPolicy } from '../core/lbug/extension-loader.js';
import { t } from './i18n/index.js';
@@ -154,22 +146,6 @@ export function pageSizeDoctorLines(
return lines;
}
/**
* The hintless buffer-pool doctor line (#2631) — the pool the next Database
* open in THIS process would get. Same plain-params testable-helper shape as
* pageSizeDoctorLines above. `pool` is getEffectiveBufferPoolSize(): `0` is
* the pass-through sentinel for LadybugDB's native 80%-of-RAM default, never
* printed as "0 MiB". `envRaw` (the raw GITNEXUS_LBUG_BUFFER_POOL_SIZE value)
* marks operator-supplied absolute values as "(env override)" — no scaling
* suffix: the hintless default is deliberately unscaled (#2557), and an env
* value is absolute, so a "×N" note would misdescribe both.
*/
export function poolSizeDoctorLine(pool: number, envRaw: string | undefined): string {
const value = pool === 0 ? 'native 80% of RAM' : `${Math.round(pool / (1024 * 1024))} MiB`;
const envNote = envRaw !== undefined && envRaw.trim().length > 0 ? ' (env override)' : '';
return ` ${padDisplayEnd('pool size', 10)}${value}${envNote}`;
}
export const doctorCommand = async () => {
const fingerprint = getRuntimeFingerprint();
const capabilities = getRuntimeCapabilities();
@@ -188,11 +164,6 @@ export const doctorCommand = async () => {
for (const line of pageSizeDoctorLines(getOsPageSize(), fingerprint.ladybugdb)) {
console.log(line);
}
// Hintless buffer pool for the next DB open (#2631). Literal label like
// the page size line above (no i18n key).
console.log(
poolSizeDoctorLine(getEffectiveBufferPoolSize(), process.env.GITNEXUS_LBUG_BUFFER_POOL_SIZE),
);
const nativeCheck = checkLbugNative();
if (nativeCheck.ok) {
console.log(` ${padDisplayEnd('native', 10)}✓ lbugjs.node loaded`);
@@ -224,32 +195,8 @@ export const doctorCommand = async () => {
console.log(` ${padDisplayEnd('', 18)}${remedy}`);
}
}
// Live LOAD probe for VECTOR too (#2623). The static capability is just
// `platform !== 'win32'`, so it printed "available" on the very machines
// where analyze was failing to load the extension — the same contradiction
// #2374 fixed for FTS above, and exactly what #2623's reporter saw while
// every incremental analyze died on an unloaded VECTOR extension.
const vectorProbe = nativeCheck.ok
? await probeVectorExtensionLoad()
: { loaded: false, reason: 'LadybugDB native module (lbugjs.node) failed to load' };
console.log(
` ${label('doctor.labels.vectorIndex', 18)}${vectorProbe.loaded ? 'available' : 'unavailable'}`,
);
if (!vectorProbe.loaded && vectorProbe.reason) {
console.log(` ${padDisplayEnd('', 18)}${vectorProbe.reason}`);
const { kind, remedy } = diagnoseExtensionLoad(vectorProbe.reason, 'VECTOR');
if (kind !== 'unknown') {
console.log(` ${padDisplayEnd('', 18)}${remedy}`);
}
}
// Semantic mode follows the probe, not the platform: without a loadable
// VECTOR extension the index can be neither built nor queried, so search is
// really on exact scan no matter what the platform would allow.
console.log(
` ${label('doctor.labels.semanticMode', 18)}${
vectorProbe.loaded ? capabilities.semanticMode : 'exact-scan'
}`,
);
console.log(` ${label('doctor.labels.vectorIndex', 18)}${capabilities.vector}`);
console.log(` ${label('doctor.labels.semanticMode', 18)}${capabilities.semanticMode}`);
// Surface the optional-extension install policy so offline users can see
// whether analyze/query will reach the network (extension.ladybugdb.com).
// Literal label (like the 'native' line) to avoid adding i18n keys.
-29
View File
@@ -3,7 +3,6 @@ import fs from 'fs/promises';
import nodePath from 'path';
import type { Path } from 'path-scurry';
import { logger } from '../core/logger.js';
import { getCoreExcludesFilePath, getGitInfoExcludePath } from '../storage/git.js';
const DEFAULT_IGNORE_LIST = new Set([
// Version Control
@@ -351,8 +350,6 @@ export const isHardcodedIgnoredDirectory = (name: string): boolean => {
export interface IgnoreOptions {
/** Skip .gitignore parsing, only read .gitnexusignore. Defaults to GITNEXUS_NO_GITIGNORE env var. */
noGitignore?: boolean;
/** Skip core.excludesFile and $GIT_COMMON_DIR/info/exclude. Defaults to GITNEXUS_NO_GLOBAL_IGNORE env var. */
noGlobalIgnore?: boolean;
}
export const loadIgnoreRules = async (
@@ -362,32 +359,6 @@ export const loadIgnoreRules = async (
const ig = ignore();
let hasRules = false;
// Mirror git's own precedence for ignore sources (gitignore(5)): patterns
// from core.excludesFile are consulted first (lowest precedence — git's
// real global, all-repos file), then $GIT_COMMON_DIR/info/exclude
// (per-repo, untracked — no write access to the repo needed), then
// .gitignore/.gitnexusignore below. Later ig.add() calls win on
// conflicting patterns, matching git's own last-match-wins semantics (#2606).
const skipGlobalIgnore = options?.noGlobalIgnore ?? !!process.env.GITNEXUS_NO_GLOBAL_IGNORE;
if (!skipGlobalIgnore) {
const globalSources = [
getCoreExcludesFilePath(repoPath),
getGitInfoExcludePath(repoPath),
].filter((candidate): candidate is string => candidate !== null);
for (const sourcePath of globalSources) {
try {
const content = await fs.readFile(sourcePath, 'utf-8');
ig.add(content);
hasRules = true;
} catch (err: unknown) {
const code = (err as NodeJS.ErrnoException).code;
if (code !== 'ENOENT') {
logger.warn(` Warning: could not read ${sourcePath}: ${(err as Error).message}`);
}
}
}
}
// Allow users to bypass .gitignore parsing (e.g. when .gitignore accidentally excludes source files)
const skipGitignore = options?.noGitignore ?? !!process.env.GITNEXUS_NO_GITIGNORE;
const filenames = skipGitignore ? ['.gitnexusignore'] : ['.gitignore', '.gitnexusignore'];
+14 -39
View File
@@ -29,7 +29,6 @@ interface HttpConfig {
maxAttempts: number;
retryCapMs: number;
minIntervalMs: number;
requestDimensions?: number;
}
export interface EmbeddingRequestOptions {
@@ -107,26 +106,20 @@ const paceHttpRequest = async (minIntervalMs: number, signal?: AbortSignal): Pro
};
/**
* Stable lead of a {@link readConfig} malformed dims-env error. `readConfig`
* throws a plain `Error` (not an {@link HttpEmbeddingError}) for a malformed
* `GITNEXUS_EMBEDDING_DIMS` or `GITNEXUS_EMBEDDING_REQUEST_DIMS` because it's a
* *config* mistake, not an endpoint failure — so the CLI recognizes it by this
* lead ({@link isHttpEmbeddingDimsError}) and prints a clean config message
* instead of a raw stack dump. Each var names itself so the message points the
* operator at the variable they actually set, not a sibling. See #2385.
* Stable lead of the {@link readConfig} malformed-`GITNEXUS_EMBEDDING_DIMS`
* error. `readConfig` throws a plain `Error` (not an {@link HttpEmbeddingError})
* because this is a *config* mistake, not an endpoint failure — so the CLI
* recognizes it by this lead ({@link isHttpEmbeddingDimsError}) and prints a
* clean config message instead of a raw stack dump. See #2385.
*/
const dimsEnvErrorLead = (name: string): string => `${name} must be a positive integer`;
const EMBEDDING_DIMS_ENV_ERROR_LEAD = dimsEnvErrorLead('GITNEXUS_EMBEDDING_DIMS');
const EMBEDDING_REQUEST_DIMS_ENV_ERROR_LEAD = dimsEnvErrorLead('GITNEXUS_EMBEDDING_REQUEST_DIMS');
const EMBEDDING_DIMS_ENV_ERROR_LEAD = 'GITNEXUS_EMBEDDING_DIMS must be a positive integer';
/**
* @internal Exported for the CLI analyze error handler. True when `message` is a
* {@link readConfig} malformed dims-env config error (a plain `Error`) — for
* either `GITNEXUS_EMBEDDING_DIMS` or `GITNEXUS_EMBEDDING_REQUEST_DIMS`.
* @internal Exported for the CLI analyze error handler. True when `message` is
* the {@link readConfig} malformed-DIMS config error (a plain `Error`).
*/
export const isHttpEmbeddingDimsError = (message: string): boolean =>
message.includes(EMBEDDING_DIMS_ENV_ERROR_LEAD) ||
message.includes(EMBEDDING_REQUEST_DIMS_ENV_ERROR_LEAD);
message.includes(EMBEDDING_DIMS_ENV_ERROR_LEAD);
/**
* Build config from the current process.env snapshot.
@@ -154,23 +147,6 @@ const readConfig = (): HttpConfig | null => {
dimensions = parsed;
}
const rawRequestDims = process.env.GITNEXUS_EMBEDDING_REQUEST_DIMS?.trim();
let requestDimensions = dimensions;
if (rawRequestDims) {
if (/^(omit|none|off|false|0)$/i.test(rawRequestDims)) {
requestDimensions = undefined;
} else {
if (!/^\d+$/.test(rawRequestDims)) {
throw new Error(`${EMBEDDING_REQUEST_DIMS_ENV_ERROR_LEAD}, got "${rawRequestDims}"`);
}
const parsed = parseInt(rawRequestDims, 10);
if (parsed <= 0) {
throw new Error(`${EMBEDDING_REQUEST_DIMS_ENV_ERROR_LEAD}, got "${rawRequestDims}"`);
}
requestDimensions = parsed;
}
}
return {
baseUrl: baseUrl.replace(/\/+$/, ''),
model,
@@ -187,7 +163,6 @@ const readConfig = (): HttpConfig | null => {
300_000,
),
minIntervalMs: parseNonNegativeIntegerEnv('GITNEXUS_EMBEDDING_MIN_INTERVAL_MS', 0, 300_000),
requestDimensions,
};
};
@@ -308,9 +283,9 @@ const isEmbeddingItem = (item: unknown): item is EmbeddingItem =>
* the `dimensions` field in the request body. Endpoints that implement
* Matryoshka truncation (OpenAI text-embedding-3-*, Cohere embed-v3,
* Voyage) return a truncated vector at that size; endpoints that do not
* recognise the field may ignore it or return 400. Set
* `GITNEXUS_EMBEDDING_REQUEST_DIMS=omit` for strict backends while keeping
* `GITNEXUS_EMBEDDING_DIMS` set to the returned vector size.
* recognise the field may ignore it or return 400. Leave
* `GITNEXUS_EMBEDDING_DIMS` unset for strict backends that reject
* unknown fields.
*/
const httpEmbedBatch = async (
url: string,
@@ -459,7 +434,7 @@ export const httpEmbed = async (
config.model,
config.apiKey,
batchIndex,
config.requestDimensions,
config.dimensions,
requestOptions,
config.maxAttempts,
config.retryCapMs,
@@ -516,7 +491,7 @@ export const httpEmbedQuery = async (
config.model,
config.apiKey,
0,
config.requestDimensions,
config.dimensions,
requestOptions,
config.maxAttempts,
config.retryCapMs,
@@ -3,10 +3,8 @@
*
* `module.registerHooks` — the synchronous ESM/CJS resolution-hook API the
* embedding-stack resolvers rely on — was added in Node 22.15.0 (and 23.5.0 on
* the 23.x line). The gitnexus engines floor is `^22.18.0 || >=24.11.0`, so
* every supported runtime exposes it — but `engines` is advisory (not
* engine-strict), so a below-floor Node (22.0–22.14, or the unsupported
* 23.0–23.4 line) can still run, where the export is absent.
* the 23.x line). The gitnexus engines floor is `>=22.0.0`, which admits Node
* 22.0–22.14 AND 23.0–23.4, where the export is absent.
*
* In this `"type": "module"` package, a *static named* import of a missing
* builtin export (`import { registerHooks } from 'node:module'`) is a
@@ -52,8 +52,8 @@
* per-resolution cost is a single string comparison.
*
* `module.registerHooks` is marked `@experimental` and requires Node >= 22.15
* (below the gitnexus engines floor of `^22.18.0 || >=24.11.0`). On below-floor
* runtimes it is absent and this is a graceful no-op: embeddings then resolve onnxruntime-common exactly
* (the gitnexus engines floor is >= 22.0.0). On older runtimes it is absent and
* this is a graceful no-op: embeddings then resolve onnxruntime-common exactly
* as before — fine on hoisted layouts. Any failure during installation is
* swallowed.
*/
@@ -100,9 +100,9 @@ export const ensureOnnxRuntimeCommonResolvable = (): void => {
attempted = true;
try {
// Node < 22.15 / < 23.5 (below the gitnexus engines floor of
// ^22.18.0 || >=24.11.0): no synchronous hooks API. Degrade gracefully —
// the import still works on hoisted layouts.
// Node < 22.15 / < 23.5 (the gitnexus engines floor is >= 22.0.0): no
// synchronous hooks API. Degrade gracefully — the import still works on
// hoisted layouts.
const registerHooks = getRegisterHooks();
if (typeof registerHooks !== 'function') return;
@@ -36,8 +36,8 @@
* So CUDA-12 hosts, Windows (DirectML), macOS, and CPU-only hosts are
* untouched. Idempotent; any failure is swallowed and leaves the default
* resolution exactly as before. `module.registerHooks` requires Node >= 22.15
* (below the gitnexus engines floor of `^22.18.0 || >=24.11.0`); on below-floor
* runtimes the redirect is a no-op, but the default copy's CUDA major is still probed so an
* (the gitnexus engines floor is >= 22.0.0); on older runtimes the redirect is
* a no-op, but the default copy's CUDA major is still probed so an
* already-matching host (e.g. CUDA 12 + transformers' CUDA-12 build) keeps
* auto-selecting the GPU.
* `npm link` / symlinked local-dev checkouts are a known caveat: `resolveOurOrtNodeDir`/
@@ -191,7 +191,7 @@ export const ensureEmbeddingStackResolvable = (): void => {
hookAttempted = true;
try {
// Node < 22.15 / < 23.5 (below the engines floor of ^22.18.0 || >=24.11.0): no synchronous hooks
// Node < 22.15 / < 23.5 (engines floor is >= 22.0.0): no synchronous hooks
// API. Degrade gracefully — normally-installed stacks still resolve; only
// the runtime-prefix fallback is unavailable. Reachable now that the import
// is a namespace access (see node-module-compat.ts) rather than a static
-2
View File
@@ -1031,8 +1031,6 @@ const LBUG_OPEN_RETRY_PATTERNS = [
'lock held by another process',
];
// Cross-repo bridge RO open retry. Catalogued as entry 5 of the lbug-config
// retry-budget registry; caps back-off so total wait ~3s.
const LBUG_OPEN_RETRY_ATTEMPTS = 10;
const LBUG_OPEN_RETRY_BASE_MS = 100;
/** Cap individual back-off delays so the total wait is bounded (~3s). */
@@ -59,7 +59,6 @@ export interface SpringBeanCandidateAdapter {
}
type OwnedTypeNamesByOwner = ReadonlyMap<string, ReadonlySet<string>>;
type RecognizedAnnotationNames = { readonly has: (value: string) => boolean };
function simpleNameOf(def: SymbolDefinition): string | undefined {
const qualifiedName = def.qualifiedName;
@@ -153,11 +152,7 @@ function hasVisibleTypeBinding(
return false;
}
function wildcardImportTarget(
parsed: ParsedFile,
simpleName: string,
recognizedAnnotations: RecognizedAnnotationNames,
): string | undefined {
function wildcardImportTarget(parsed: ParsedFile, simpleName: string): string | undefined {
const wildcardPackages = new Set(
parsed.parsedImports
.filter((entry) => entry.kind === 'wildcard')
@@ -166,40 +161,37 @@ function wildcardImportTarget(
if (wildcardPackages.size !== 1) return undefined;
const [packageName] = wildcardPackages;
const target = `${packageName}.${simpleName}`;
return recognizedAnnotations.has(target) ? target : undefined;
return SPRING_BEAN_STEREOTYPES.has(target) ? target : undefined;
}
/** Build a scope-aware Spring annotation resolver shared by framework hooks. */
export function createSpringAnnotationNameResolver(indexes: ScopeResolutionIndexes) {
const ownedTypeNamesByOwner = buildOwnedTypeNamesByOwner(indexes);
return (
rawName: string,
parsed: ParsedFile,
enclosingScope: ScopeId | null,
recognizedAnnotations: RecognizedAnnotationNames,
isPackageVisibilityIncomplete: boolean,
): string | undefined => {
if (rawName.includes('.')) {
return recognizedAnnotations.has(rawName) ? rawName : undefined;
}
function resolveSpringAnnotation(
rawName: string,
parsed: ParsedFile,
enclosingScope: ScopeId | null,
indexes: ScopeResolutionIndexes,
ownedTypeNamesByOwner: OwnedTypeNamesByOwner,
isPackageVisibilityIncomplete: boolean,
): string | undefined {
if (rawName.includes('.')) {
return SPRING_BEAN_STEREOTYPES.has(rawName) ? rawName : undefined;
}
if (hasLexicalTypeDeclaration(enclosingScope, rawName, indexes)) return undefined;
if (hasInheritedTypeDeclaration(enclosingScope, rawName, indexes, ownedTypeNamesByOwner)) {
return undefined;
}
if (hasLexicalTypeDeclaration(enclosingScope, rawName, indexes)) return undefined;
if (hasInheritedTypeDeclaration(enclosingScope, rawName, indexes, ownedTypeNamesByOwner)) {
return undefined;
}
const explicitImports = explicitImportTargets(parsed, rawName);
if (explicitImports.size > 0) {
if (explicitImports.size !== 1) return undefined;
const [imported] = explicitImports;
return recognizedAnnotations.has(imported) ? imported : undefined;
}
const explicitImports = explicitImportTargets(parsed, rawName);
if (explicitImports.size > 0) {
if (explicitImports.size !== 1) return undefined;
const [imported] = explicitImports;
return SPRING_BEAN_STEREOTYPES.has(imported) ? imported : undefined;
}
const wildcardTarget = wildcardImportTarget(parsed, rawName, recognizedAnnotations);
if (wildcardTarget === undefined || isPackageVisibilityIncomplete) return undefined;
const wildcardTarget = wildcardImportTarget(parsed, rawName);
if (wildcardTarget === undefined || isPackageVisibilityIncomplete) return undefined;
return hasVisibleTypeBinding(enclosingScope, rawName, indexes) ? undefined : wildcardTarget;
};
return hasVisibleTypeBinding(enclosingScope, rawName, indexes) ? undefined : wildcardTarget;
}
/** Build a language hook that enriches Class nodes after scope resolution. */
@@ -210,7 +202,7 @@ export function createSpringBeanCandidateAttacher(adapter: SpringBeanCandidateAd
nodeLookup: GraphNodeLookup,
indexes: ScopeResolutionIndexes,
): void => {
const resolveSpringAnnotation = createSpringAnnotationNameResolver(indexes);
const ownedTypeNamesByOwner = buildOwnedTypeNamesByOwner(indexes);
for (const parsed of parsedFiles) {
for (const fact of adapter.getClassAnnotationFacts(parsed.filePath)) {
const classScope = indexes.scopeTree.getScope(fact.classScopeId);
@@ -229,7 +221,8 @@ export function createSpringBeanCandidateAttacher(adapter: SpringBeanCandidateAd
rawName,
parsed,
classScope.parent,
SPRING_BEAN_STEREOTYPES,
indexes,
ownedTypeNamesByOwner,
adapter.isPackageVisibilityIncomplete(parsed.filePath),
);
if (annotation !== undefined) recognized.add(annotation);
@@ -1,166 +0,0 @@
import type { GraphNode } from 'gitnexus-shared';
import type { KnowledgeGraph } from '../../../graph/types.js';
import { generateId } from '../../../../lib/utils.js';
export const SPRING_CONFIG_DESCRIPTION = 'Spring configuration property';
export interface SpringValueConsumer {
readonly kind: 'value';
readonly fieldName: string;
readonly line: number;
readonly keys: readonly string[];
}
export interface SpringConfigurationPropertiesConsumer {
readonly kind: 'configuration-properties';
readonly className: string;
readonly line: number;
readonly prefix: string;
}
export type SpringConfigConsumer = SpringValueConsumer | SpringConfigurationPropertiesConsumer;
export interface SpringConfigConsumerBatch {
readonly filePath: string;
readonly consumers: readonly SpringConfigConsumer[];
}
function closestNode(
candidates: readonly GraphNode[],
filePath: string,
name: string,
line: number,
): GraphNode | undefined {
return candidates
.filter((node) => node.properties.filePath === filePath && node.properties.name === name)
.sort(
(left, right) =>
Math.abs(Number(left.properties.startLine ?? 0) - line) -
Math.abs(Number(right.properties.startLine ?? 0) - line),
)[0];
}
function markUnresolved(node: GraphNode, key: string): void {
const marker = `Spring config unresolved: ${key}`;
const existing =
typeof node.properties.description === 'string' ? node.properties.description : '';
if (existing.includes(marker)) return;
node.properties.description = existing.length > 0 ? `${existing}; ${marker}` : marker;
}
function relaxedName(value: string): string {
return value.toLowerCase().replace(/[-_.]/g, '');
}
function isSpringConfigNode(node: GraphNode): boolean {
return (
node.label === 'Property' &&
typeof node.properties.description === 'string' &&
node.properties.description.startsWith(SPRING_CONFIG_DESCRIPTION)
);
}
/**
* Attach normalized, language-provider-produced Spring consumers to config
* keys already present in the shared graph.
*/
export function bindSpringConfigConsumers(
graph: KnowledgeGraph,
batches: readonly SpringConfigConsumerBatch[],
): void {
if (batches.length === 0) return;
const configNodes: GraphNode[] = [];
const propertyNodes: GraphNode[] = [];
const classNodes: GraphNode[] = [];
for (const node of graph.iterNodes()) {
if (isSpringConfigNode(node)) configNodes.push(node);
else if (node.label === 'Property') propertyNodes.push(node);
else if (node.label === 'Class' || node.label === 'Record') classNodes.push(node);
}
const keyNodes = new Map<string, GraphNode[]>();
for (const node of configNodes) {
const key = String(node.properties.name);
const bucket = keyNodes.get(key) ?? [];
bucket.push(node);
keyNodes.set(key, bucket);
}
const propertiesByOwner = new Map<string, GraphNode[]>();
for (const rel of graph.iterRelationshipsByType('HAS_PROPERTY')) {
const property = graph.getNode(rel.targetId);
if (property?.label !== 'Property' || isSpringConfigNode(property)) continue;
const members = propertiesByOwner.get(rel.sourceId) ?? [];
members.push(property);
propertiesByOwner.set(rel.sourceId, members);
}
const addBinding = (
source: GraphNode,
target: GraphNode,
reason: string,
confidence: number,
): void => {
const edgeId = generateId('USES', `${source.id}->${target.id}:${reason}`);
graph.addRelationship({
id: edgeId,
sourceId: source.id,
targetId: target.id,
type: 'USES',
confidence,
reason,
});
};
for (const { filePath, consumers } of batches) {
for (const consumer of consumers) {
if (consumer.kind === 'value') {
const field = closestNode(propertyNodes, filePath, consumer.fieldName, consumer.line);
if (field === undefined) continue;
for (const key of consumer.keys) {
const matches = keyNodes.get(key) ?? [];
if (matches.length === 0) {
markUnresolved(field, key);
continue;
}
for (const match of matches) {
addBinding(field, match, `spring-config:@Value ${key}`, 1);
}
}
continue;
}
const owner = closestNode(classNodes, filePath, consumer.className, consumer.line);
if (owner === undefined) continue;
const prefix = `${consumer.prefix}.`;
const matches = configNodes.filter((node) => {
const key = String(node.properties.name);
return key === consumer.prefix || key.startsWith(prefix);
});
if (matches.length === 0) {
markUnresolved(owner, consumer.prefix);
continue;
}
for (const match of matches) {
addBinding(owner, match, `spring-config:@ConfigurationProperties ${consumer.prefix}`, 0.95);
}
for (const field of propertiesByOwner.get(owner.id) ?? []) {
const fieldName = relaxedName(String(field.properties.name));
for (const match of matches) {
const key = String(match.properties.name);
const suffix = key === consumer.prefix ? '' : key.slice(prefix.length);
const firstSegment = suffix.split(/[.\[]/, 1)[0];
if (firstSegment.length === 0 || relaxedName(firstSegment) !== fieldName) continue;
addBinding(
field,
match,
`spring-config:@ConfigurationProperties field ${consumer.prefix}`,
0.95,
);
}
}
}
}
}
@@ -1,19 +0,0 @@
import type { AnalysisFeatureDescriptor } from '../../../analysis-features.js';
function isSpringApplicationConfig(filePath: string): boolean {
const base = filePath.replaceAll('\\', '/').split('/').pop() ?? '';
return /^application(?:-[^.]+)?\.(?:properties|ya?ml)$/i.test(base);
}
/** Durable completeness contract for Java Spring configuration bindings. */
export const SPRING_CONFIG_BINDINGS_FEATURE: AnalysisFeatureDescriptor = {
id: 'spring.config-bindings',
version: 1,
// Java sources need consumer extraction even without config files (missing
// placeholders still get unresolved markers). Config-only repositories also
// need a one-time rebuild to backfill language-agnostic Property nodes.
appliesTo: (filePaths) =>
filePaths.some(
(filePath) => filePath.toLowerCase().endsWith('.java') || isSpringApplicationConfig(filePath),
),
};
@@ -9,7 +9,6 @@ import {
type JvmPackageFact,
} from '../jvm/package-facts.js';
import { getJavaPackageFact, setJavaPackageFact } from './package-facts.js';
import type { JavaSpringConfigConsumerFact } from './spring-config-bindings.js';
export type JavaClassAnnotationFact = ClassAnnotationFact;
@@ -17,16 +16,13 @@ export interface JavaCaptureSideChannel {
readonly kind: 'java';
readonly packageFact: JvmPackageFact;
readonly classAnnotations: readonly JavaClassAnnotationFact[];
readonly springConfigConsumers?: readonly JavaSpringConfigConsumerFact[];
}
const classAnnotations = createClassAnnotationFactStore();
const springConfigConsumers = new Map<string, readonly JavaSpringConfigConsumerFact[]>();
/** Clear facts retained by a prior workspace pass in a long-lived process. */
export function clearJavaClassAnnotationFacts(): void {
classAnnotations.clear();
springConfigConsumers.clear();
}
/** Store the annotation syntax collected by Java's existing scope-query traversal. */
@@ -37,35 +33,17 @@ export function setJavaClassAnnotationFacts(
classAnnotations.set(filePath, facts);
}
export function setJavaSpringConfigConsumerFacts(
filePath: string,
facts: readonly JavaSpringConfigConsumerFact[],
): void {
if (facts.length === 0) springConfigConsumers.delete(filePath);
else springConfigConsumers.set(filePath, facts);
}
export function getJavaSpringConfigConsumerFacts(
filePath: string,
): readonly JavaSpringConfigConsumerFact[] {
return springConfigConsumers.get(filePath) ?? [];
}
/** Snapshot worker-local Java annotation facts for ParsedFile serialization. */
export function collectJavaCaptureSideChannel(
filePath: string,
): JavaCaptureSideChannel | undefined {
const facts = classAnnotations.get(filePath);
const configConsumers = springConfigConsumers.get(filePath) ?? [];
const packageFact = getJavaPackageFact(filePath);
if (facts.length === 0 && configConsumers.length === 0 && packageFact === undefined) {
return undefined;
}
if (facts.length === 0 && packageFact === undefined) return undefined;
return {
kind: 'java',
packageFact: packageFact ?? UNKNOWN_JVM_PACKAGE_FACT,
classAnnotations: facts,
...(configConsumers.length > 0 ? { springConfigConsumers: configConsumers } : {}),
};
}
@@ -84,15 +62,10 @@ export function applyJavaCaptureSideChannel(parsed: ParsedFile): void {
!Array.isArray(data.classAnnotations)
) {
setJavaClassAnnotationFacts(parsed.filePath, []);
setJavaSpringConfigConsumerFacts(parsed.filePath, []);
setJavaPackageFact(parsed.filePath, UNKNOWN_JVM_PACKAGE_FACT);
return;
}
setJavaClassAnnotationFacts(parsed.filePath, data.classAnnotations);
setJavaSpringConfigConsumerFacts(
parsed.filePath,
Array.isArray(data.springConfigConsumers) ? data.springConfigConsumers : [],
);
setJavaPackageFact(
parsed.filePath,
isJvmPackageFact(data.packageFact) ? data.packageFact : UNKNOWN_JVM_PACKAGE_FACT,
@@ -32,13 +32,9 @@ import { getJavaParser, getJavaScopeQuery } from './query.js';
import { recordCacheHit, recordCacheMiss } from './cache-stats.js';
import { getTreeSitterBufferSize } from '../../constants.js';
import { parseSourceSafe } from '../../../tree-sitter/safe-parse.js';
import {
setJavaClassAnnotationFacts,
setJavaSpringConfigConsumerFacts,
} from './capture-side-channel.js';
import { setJavaClassAnnotationFacts } from './capture-side-channel.js';
import { captureJavaPackageFact } from './package-facts.js';
import { synthesizeCallableFlowCaptures } from '../../utils/callable-flow-captures.js';
import { captureJavaSpringConfigConsumerFacts } from './spring-config-bindings.js';
/** Declaration anchors that carry function-like arity metadata. */
const FUNCTION_DECL_TAGS = ['@declaration.method', '@declaration.constructor'] as const;
@@ -154,29 +150,6 @@ export function emitJavaScopeCaptures(
continue;
}
// Normalize a `new`-expression receiver to its constructed type's simple
// name: `new Local().inner()` binds the WHOLE `object_creation_expression`
// as `@reference.receiver`, so its raw text is `"new Local()"` — a string
// that can never match a scope binding, so the compound-receiver resolver
// silently falls through to name-only fallback resolution and picks the
// wrong same-named method on a collision (#2564). Rewriting the text to
// just `Local` lets Case 2 (class-name / static receiver) in
// receiver-bound-calls.ts resolve it via its normal MRO walk. Mirrors the
// established `normalizePhpReceiver` precedent (php/captures.ts) — a
// language-local capture rewrite, no shared-pipeline change.
if (grouped['@reference.receiver'] !== undefined) {
const receiverNode = nodeIfType(nodeMap['@reference.receiver'], 'object_creation_expression');
const typeNode = receiverNode?.childForFieldName('type');
const simpleName = typeNode ? javaBaseSimpleNameOf(typeNode) : undefined;
if (simpleName !== undefined) {
grouped['@reference.receiver'] = syntheticCapture(
'@reference.receiver',
receiverNode!,
simpleName,
);
}
}
// Filter read.member when it's a child of method_invocation or assignment.
// `@reference.read.member` is captured directly on the `field_access` node.
if (grouped['@reference.read.member'] !== undefined) {
@@ -284,10 +257,6 @@ export function emitJavaScopeCaptures(
}
setJavaClassAnnotationFacts(filePath, materializeClassAnnotationFacts(classAnnotations));
setJavaSpringConfigConsumerFacts(
filePath,
captureJavaSpringConfigConsumerFacts(tree.rootNode, filePath),
);
return [
...resolveVarTypeBindings(out),
@@ -372,49 +341,23 @@ function synthesizeJavaAnonymousClassDeclarations(rootNode: SyntaxNode): Capture
// constant's class extends its HOST ENUM (javac semantics), so the
// inherits reference names the enum — giving `mroFor(E$N) ∋ E` and
// keeping bare calls from the body to the enum's own helpers alive
// through the ownership gate's MRO arm.
// through the ownership gate's MRO arm. No receiver typeBinding piece:
// constants are not variable initializers; `E.A.hook()` dispatch rides
// the existing enum receiver machinery.
for (const constant of rootNode.descendantsOfType('enum_constant')) {
const name = synthesizeJavaAnonymousClassName(constant);
if (name === undefined) continue;
const body = constant.childForFieldName?.('body');
if (body === null || body === undefined || body.type !== 'class_body') continue;
out.push({
'@declaration.class': nodeToCapture('@declaration.class', body),
'@declaration.name': syntheticCapture('@declaration.name', body, name),
});
const hostEnum = javaEnclosingEnumNameOf(constant);
const bodyNode = constant.childForFieldName?.('body');
const isBodied = bodyNode !== null && bodyNode !== undefined && bodyNode.type === 'class_body';
const bodiedName = synthesizeJavaAnonymousClassName(constant);
if (bodiedName !== undefined && isBodied) {
if (hostEnum !== undefined) {
out.push({
'@declaration.class': nodeToCapture('@declaration.class', bodyNode),
'@declaration.name': syntheticCapture('@declaration.name', bodyNode, bodiedName),
});
if (hostEnum !== undefined) {
out.push({
'@reference.inherits': nodeToCapture('@reference.inherits', bodyNode),
'@reference.name': syntheticCapture('@reference.name', bodyNode, hostEnum),
});
}
}
// Receiver dispatch (#2561): `E.CONST.method()` resolves through the
// generic compound-receiver chain walk, which looks up each dotted
// segment via the owning class scope's `typeBindings` map — the same
// mechanism a field declaration uses (`private User user;` binds
// `user` on the class scope). Binding the constant's own simple name
// there — to its synthesized `E$N` class when bodied (MRO includes E,
// so members inherited from the enum still resolve), or to the host
// enum itself when body-less — makes `E.CONST.method()` resolve with
// no changes to the shared receiver-binding machinery.
//
// A bodied constant binds ONLY to its `E$N` class, never the host enum:
// if name synthesis fails on a malformed/error-recovery tree (`bodiedName`
// undefined despite a real body), emit nothing rather than silently
// misattributing an OVERRIDING constant's receiver to the enum's own
// (non-overridden) method — a wrong edge is worse than no edge. Mirrors
// the `object_creation_expression` branch, which skips on synthesis
// failure. `hostEnum` is used only for genuinely body-less constants.
const constantNameNode = constant.childForFieldName?.('name');
const constantType = isBodied ? bodiedName : hostEnum;
if (constantNameNode !== null && constantNameNode !== undefined && constantType !== undefined) {
out.push({
'@type-binding.annotation': nodeToCapture('@type-binding.annotation', constant),
'@type-binding.name': nodeToCapture('@type-binding.name', constantNameNode),
'@type-binding.type': syntheticCapture('@type-binding.type', constant, constantType),
'@reference.inherits': nodeToCapture('@reference.inherits', body),
'@reference.name': syntheticCapture('@reference.name', body, hostEnum),
});
}
}
@@ -30,7 +30,6 @@ import {
} from './index.js';
import { populateJavaPackageSiblings } from './package-siblings.js';
import { attachSpringBeanCandidateMetadata } from './spring-bean-metadata.js';
import { attachJavaSpringConfigBindings } from './spring-config-bindings.js';
import {
applyJavaCaptureSideChannel,
clearJavaClassAnnotationFacts,
@@ -84,10 +83,7 @@ const javaScopeResolver: ScopeResolver = {
populateNamespaceSiblings: populateJavaPackageSiblings,
populateRangeBindings: populateJavaCrossFileReturnTypes,
emitPostResolutionEdges: (graph, parsedFiles, nodeLookup, indexes, ctx) => {
attachSpringBeanCandidateMetadata(graph, parsedFiles, nodeLookup, indexes);
attachJavaSpringConfigBindings(graph, parsedFiles, nodeLookup, indexes, ctx);
},
emitPostResolutionEdges: attachSpringBeanCandidateMetadata,
};
export { javaScopeResolver };
@@ -1,267 +0,0 @@
import type { KnowledgeGraph } from '../../../graph/types.js';
import type { GraphNodeLookup } from '../../scope-resolution/graph-bridge/node-lookup.js';
import type { ScopeResolutionIndexes } from '../../model/scope-resolution-indexes.js';
import { makeScopeId, type ParsedFile, type ScopeId } from 'gitnexus-shared';
import {
bindSpringConfigConsumers,
type SpringConfigConsumer,
} from '../../frameworks/spring/config-bindings.js';
import { createSpringAnnotationNameResolver } from '../../frameworks/spring/bean-candidates.js';
import { parseSourceSafe } from '../../../tree-sitter/safe-parse.js';
import { nodeToCapture, type SyntaxNode } from '../../utils/ast-helpers.js';
import { getJavaParser } from './query.js';
import { getJavaSpringConfigConsumerFacts } from './capture-side-channel.js';
import { isJavaPackageSiblingVisibilityIncomplete } from './package-siblings.js';
const VALUE_ANNOTATION = 'org.springframework.beans.factory.annotation.Value';
const CONFIGURATION_PROPERTIES_ANNOTATION =
'org.springframework.boot.context.properties.ConfigurationProperties';
interface JavaAnnotation {
readonly name: string;
readonly node: SyntaxNode;
}
interface JavaImports {
readonly exact: ReadonlySet<string>;
readonly wildcard: ReadonlySet<string>;
readonly localTypes: ReadonlySet<string>;
}
export interface JavaSpringConfigConsumerFact {
readonly consumer: SpringConfigConsumer;
readonly annotationName: string;
readonly classScopeId: ScopeId;
}
function collectJavaImports(root: SyntaxNode): JavaImports {
const exact = new Set<string>();
const wildcard = new Set<string>();
const localTypes = new Set<string>();
for (const node of root.descendantsOfType('import_declaration')) {
const imported = node.text
.replace(/^\s*import\s+(?:static\s+)?/, '')
.replace(/;\s*$/, '')
.trim();
if (imported.endsWith('.*')) wildcard.add(imported.slice(0, -2));
else exact.add(imported);
}
for (const type of [
'class_declaration',
'interface_declaration',
'enum_declaration',
'record_declaration',
'annotation_type_declaration',
]) {
for (const node of root.descendantsOfType(type)) {
const name = node.childForFieldName('name')?.text;
if (name) localTypes.add(name);
}
}
return { exact, wildcard, localTypes };
}
function annotationsOn(node: SyntaxNode): JavaAnnotation[] {
const modifiers = node.namedChildren.find((child) => child.type === 'modifiers');
if (modifiers === undefined) return [];
const annotations: JavaAnnotation[] = [];
for (const child of modifiers.namedChildren) {
if (child.type !== 'annotation' && child.type !== 'marker_annotation') continue;
const name = child.childForFieldName('name')?.text ?? child.firstNamedChild?.text;
if (name) annotations.push({ name, node: child });
}
return annotations;
}
function resolvesToAnnotation(
rawName: string,
canonicalName: string,
imports: JavaImports,
): boolean {
if (rawName.includes('.')) return rawName === canonicalName;
if (imports.localTypes.has(rawName)) return false;
if (imports.exact.has(canonicalName)) return true;
const packageName = canonicalName.slice(0, canonicalName.lastIndexOf('.'));
return imports.wildcard.has(packageName);
}
function decodeJavaStringLiteral(literal: string): string {
const delimiterLength = literal.startsWith('"""') && literal.endsWith('"""') ? 3 : 1;
return literal
.slice(delimiterLength, -delimiterLength)
.replace(/\\u([0-9a-fA-F]{4})/g, (_match, hex: string) =>
String.fromCharCode(Number.parseInt(hex, 16)),
)
.replace(/\\(["'\\btnfr])/g, (_match, escaped: string) => {
const controls: Record<string, string> = {
b: '\b',
t: '\t',
n: '\n',
f: '\f',
r: '\r',
};
return controls[escaped] ?? escaped;
});
}
function javaStringLiterals(annotation: SyntaxNode): string[] {
return annotation
.descendantsOfType('string_literal')
.map((literal) => decodeJavaStringLiteral(literal.text));
}
/** Extract statically readable Spring placeholder keys from a Java annotation. */
export function parseValuePlaceholderKeys(annotation: SyntaxNode): string[] {
const keys = new Set<string>();
for (const literal of javaStringLiterals(annotation)) {
for (const match of literal.matchAll(/\$\{([^{}]+)\}/g)) {
const key = match[1].split(':', 1)[0].trim();
if (/^[A-Za-z0-9_.-]+$/.test(key)) keys.add(key);
}
}
return [...keys];
}
/** Extract `prefix`/`value` (or the positional value) from the annotation. */
export function parseConfigurationPropertiesPrefix(annotation: SyntaxNode): string | null {
const named = annotation.descendantsOfType('element_value_pair').find((pair) => {
const key = pair.childForFieldName('key')?.text;
return key === 'prefix' || key === 'value';
});
const namedValue = named?.childForFieldName('value');
const argumentsNode = annotation.childForFieldName('arguments');
const literalNode =
(namedValue?.type === 'string_literal'
? namedValue
: namedValue?.descendantsOfType('string_literal')[0]) ??
(named === undefined
? argumentsNode?.namedChildren.find((child) => child.type === 'string_literal')
: undefined);
if (literalNode === undefined) return null;
const prefix = decodeJavaStringLiteral(literalNode.text)
.trim()
.replace(/^\.+|\.+$/g, '');
return /^[A-Za-z0-9_.-]+$/.test(prefix) ? prefix : null;
}
function classScopeId(filePath: string, declaration: SyntaxNode): ScopeId {
return makeScopeId({
filePath,
range: nodeToCapture('@scope.class', declaration).range,
kind: 'Class',
});
}
function enclosingClass(node: SyntaxNode): SyntaxNode | undefined {
let current = node.parent;
while (current !== null) {
if (current.type === 'class_declaration' || current.type === 'record_declaration') {
return current;
}
current = current.parent;
}
return undefined;
}
/** Collect config facts from the Java parser's existing AST (no reparse). */
export function captureJavaSpringConfigConsumerFacts(
root: SyntaxNode,
filePath: string,
): JavaSpringConfigConsumerFact[] {
const imports = collectJavaImports(root);
const facts: JavaSpringConfigConsumerFact[] = [];
for (const field of root.descendantsOfType('field_declaration')) {
const annotations = annotationsOn(field).filter((annotation) =>
resolvesToAnnotation(annotation.name, VALUE_ANNOTATION, imports),
);
if (annotations.length === 0) continue;
const owner = enclosingClass(field);
if (owner === undefined) continue;
for (const declarator of field.namedChildren.filter(
(child) => child.type === 'variable_declarator',
)) {
const fieldName = declarator.childForFieldName('name')?.text;
if (!fieldName) continue;
for (const annotation of annotations) {
const keys = parseValuePlaceholderKeys(annotation.node);
if (keys.length > 0) {
facts.push({
consumer: { kind: 'value', fieldName, line: field.startPosition.row + 1, keys },
annotationName: annotation.name,
classScopeId: classScopeId(filePath, owner),
});
}
}
}
}
for (const type of ['class_declaration', 'record_declaration']) {
for (const declaration of root.descendantsOfType(type)) {
const className = declaration.childForFieldName('name')?.text;
if (!className) continue;
for (const annotation of annotationsOn(declaration)) {
if (!resolvesToAnnotation(annotation.name, CONFIGURATION_PROPERTIES_ANNOTATION, imports)) {
continue;
}
const prefix = parseConfigurationPropertiesPrefix(annotation.node);
if (prefix !== null) {
facts.push({
consumer: {
kind: 'configuration-properties',
className,
line: declaration.startPosition.row + 1,
prefix,
},
annotationName: annotation.name,
classScopeId: classScopeId(filePath, declaration),
});
}
}
}
}
return facts;
}
/** Parse Java consumers for focused unit tests; production reuses the worker AST. */
export function extractJavaSpringConfigConsumers(source: string): SpringConfigConsumer[] {
const tree = parseSourceSafe(getJavaParser(), source);
return captureJavaSpringConfigConsumerFacts(tree.rootNode, '<memory>').map(
(fact) => fact.consumer,
);
}
/** Java ScopeResolver post-resolution hook for Spring configuration consumers. */
export function attachJavaSpringConfigBindings(
graph: KnowledgeGraph,
parsedFiles: readonly ParsedFile[],
_nodeLookup: GraphNodeLookup,
indexes: ScopeResolutionIndexes,
_ctx: { readonly fileContents: ReadonlyMap<string, string> },
): void {
const resolveAnnotation = createSpringAnnotationNameResolver(indexes);
const recognizedAnnotations = new Set([VALUE_ANNOTATION, CONFIGURATION_PROPERTIES_ANNOTATION]);
const batches: Array<{ filePath: string; consumers: SpringConfigConsumer[] }> = [];
for (const parsed of parsedFiles) {
const consumers: SpringConfigConsumer[] = [];
for (const fact of getJavaSpringConfigConsumerFacts(parsed.filePath)) {
const classScope = indexes.scopeTree.getScope(fact.classScopeId);
if (classScope === undefined || classScope.kind !== 'Class') continue;
const expectedAnnotation =
fact.consumer.kind === 'value' ? VALUE_ANNOTATION : CONFIGURATION_PROPERTIES_ANNOTATION;
const enclosingScope = fact.consumer.kind === 'value' ? classScope.id : classScope.parent;
const resolved = resolveAnnotation(
fact.annotationName,
parsed,
enclosingScope,
recognizedAnnotations,
isJavaPackageSiblingVisibilityIncomplete(parsed.filePath),
);
if (resolved === expectedAnnotation) consumers.push(fact.consumer);
}
if (consumers.length > 0) batches.push({ filePath: parsed.filePath, consumers });
}
bindSpringConfigConsumers(graph, batches);
}
@@ -8,9 +8,10 @@
* 1. **Per-name import statements** — `import a, b` and
* `from m import x, y` decompose to one match per imported name
* (see `import-decomposer.ts`).
* 2. **Receiver type bindings** — methods emit an implicit `self` / `cls`
* binding, and `__init__` assignments from annotated parameters emit
* class-scoped instance-field bindings (see `receiver-binding.ts`).
* 2. **Receiver type bindings** — each `function_definition` inside a
* class body emits a `@type-binding.self` (or `@type-binding.cls`
* for `@classmethod`) capture so Pass-4 attaches the implicit
* receiver (see `receiver-binding.ts`).
*
* Pure given the input source text. No I/O, no globals consulted.
*/
@@ -24,10 +25,7 @@ import {
} from '../../utils/ast-helpers.js';
import { splitImportStatement } from './import-decomposer.js';
import { getPythonParser, getPythonScopeQuery } from './query.js';
import {
synthesizeConstructorFieldTypeBindings,
synthesizeReceiverTypeBinding,
} from './receiver-binding.js';
import { synthesizeReceiverTypeBinding } from './receiver-binding.js';
import { synthesizeDependsReferences } from './depends-references.js';
import { computePythonArityMetadata } from './arity-metadata.js';
import { recordCacheHit, recordCacheMiss } from './cache-stats.js';
@@ -135,7 +133,6 @@ export function emitPythonScopeCaptures(
if (fnNode !== null) {
const synth = synthesizeReceiverTypeBinding(fnNode);
if (synth !== null) out.push(synth);
out.push(...synthesizeConstructorFieldTypeBindings(fnNode));
for (const depRef of synthesizeDependsReferences(fnNode)) out.push(depRef);
}
continue;
@@ -119,10 +119,7 @@ export function interpretPythonTypeBinding(captures: CaptureMatch): ParsedTypeBi
// `cls` is a self-like receiver; share the source label so downstream
// `Registry.lookup` Step 2 treats them identically.
else if (captures['@type-binding.cls'] !== undefined) source = 'self';
else if (captures['@type-binding.instance-field'] !== undefined) {
source =
captures['@type-binding.parameter'] !== undefined ? 'parameter-annotation' : 'annotation';
} else if (captures['@type-binding.constructor'] !== undefined) source = 'constructor-inferred';
else if (captures['@type-binding.constructor'] !== undefined) source = 'constructor-inferred';
else if (captures['@type-binding.annotation'] !== undefined) source = 'annotation';
else if (captures['@type-binding.alias'] !== undefined) source = 'assignment-inferred';
else if (captures['@type-binding.return'] !== undefined) source = 'return-annotation';
@@ -1,6 +1,6 @@
/**
* Synthesize implicit receiver and constructor-assigned field type bindings
* for methods.
* Synthesize `@type-binding.self` / `@type-binding.cls` captures for
* methods.
*
* Tree-sitter can't easily express "the first parameter of a function
* defined directly inside a class body" via a single static query.
@@ -113,114 +113,3 @@ export function synthesizeReceiverTypeBinding(fnNode: SyntaxNode): CaptureMatch
'@type-binding.type': syntheticCapture('@type-binding.type', first, className),
};
}
/**
* Synthesize class-scope field bindings for the common Python constructor
* injection pattern:
*
* def __init__(self, service: Service):
* self.service = service
*
* An explicit field annotation (`self.service: Service = ...`) is also
* accepted and takes precedence over a parameter annotation. Deliberately do
* not infer from arbitrary unannotated RHS expressions: the receiver resolver
* needs a declared type, not a name-only guess.
*/
export function synthesizeConstructorFieldTypeBindings(fnNode: SyntaxNode): CaptureMatch[] {
if (fnNode.childForFieldName('name')?.text !== '__init__') return [];
if (findEnclosingClassDefinition(fnNode) === null) return [];
if (hasDecorator(fnNode, 'staticmethod') || hasDecorator(fnNode, 'classmethod')) return [];
const receiver = synthesizeReceiverTypeBinding(fnNode);
const receiverName = receiver?.['@type-binding.self']?.text;
if (receiverName === undefined) return [];
const parameters = fnNode.childForFieldName('parameters');
const body = fnNode.childForFieldName('body');
if (parameters === null || body === null) return [];
const parameterTypes = new Map<string, string>();
for (let i = 0; i < parameters.namedChildCount; i++) {
const parameter = parameters.namedChild(i);
if (parameter === null) continue;
const name = firstParameterName(parameter);
const annotation = parameter.childForFieldName('type');
if (name !== null && annotation !== null) parameterTypes.set(name, annotation.text);
}
type Candidate = { readonly match: CaptureMatch; readonly explicit: boolean };
const candidates = new Map<string, Candidate>();
const stack: SyntaxNode[] = [body];
while (stack.length > 0) {
const node = stack.pop()!;
if (
node !== body &&
(node.type === 'function_definition' ||
node.type === 'lambda' ||
node.type === 'class_definition' ||
node.type === 'if_statement' ||
node.type === 'for_statement' ||
node.type === 'while_statement' ||
node.type === 'try_statement' ||
node.type === 'match_statement')
) {
continue;
}
if (node.type === 'assignment') {
const left = node.childForFieldName('left');
const right = node.childForFieldName('right');
if (left?.type === 'attribute') {
const object = left.childForFieldName('object');
const field = left.childForFieldName('attribute');
if (object?.type === 'identifier' && object.text === receiverName && field !== null) {
const explicitType = node.childForFieldName('type');
const parameterType =
right?.type === 'identifier' ? parameterTypes.get(right.text) : undefined;
const typeName = explicitType?.text ?? parameterType;
if (typeName !== undefined) {
const explicit = explicitType !== null;
const existing = candidates.get(field.text);
if (existing === undefined || explicit || !existing.explicit) {
candidates.set(field.text, {
explicit,
match: {
'@type-binding.name': syntheticCapture('@type-binding.name', field, field.text),
'@type-binding.type': syntheticCapture(
'@type-binding.type',
explicitType ?? right ?? field,
typeName,
),
...(explicit
? {}
: {
'@type-binding.parameter': syntheticCapture(
'@type-binding.parameter',
right ?? field,
'1',
),
}),
'@type-binding.instance-field': syntheticCapture(
'@type-binding.instance-field',
node,
'1',
),
},
});
}
}
}
}
}
// Push in reverse so the LIFO walk visits source order. That keeps Map
// insertion order (and therefore emitted capture order) deterministic.
for (let i = node.namedChildCount - 1; i >= 0; i--) {
const child = node.namedChild(i);
if (child !== null) stack.push(child);
}
}
return [...candidates.values()].map(({ match }) => match);
}
@@ -36,23 +36,15 @@ export function pythonFunctionDefinitionLabel(
// ─── bindingScopeFor ──────────────────────────────────────────────────────
/** Python has no block scope, so the central extractor's "innermost
* enclosing scope" default is already correct for ordinary bindings.
* Constructor-injected instance fields are the exception: their marker is
* anchored inside `__init__`, but compound receiver resolution needs the
* field type on the enclosing Class scope. */
* enclosing scope" default is already correct: `for x in …` creates
* `x` in the enclosing function/module scope (because we never emit a
* `@scope.block` for the for-loop body), comprehension variables stay
* in their expression context, etc. Returns `null` to delegate. */
export function pythonBindingScopeFor(
decl: CaptureMatch,
innermost: Scope,
tree: ScopeTree,
_decl: CaptureMatch,
_innermost: Scope,
_tree: ScopeTree,
): ScopeId | null {
if (decl['@type-binding.instance-field'] !== undefined) {
let current: Scope | undefined = innermost;
while (current !== undefined) {
if (current.kind === 'Class') return current.id;
if (current.parent === null) break;
current = tree.getScope(current.parent);
}
}
return null;
}
@@ -2,23 +2,8 @@ import type { CaptureMatch, ParsedImport, ParsedTypeBinding, TypeRef } from 'git
const REF_PREFIX_RE = /^&\s*(mut\s+)?/;
const PTR_PREFIX_RE = /^\*\s*(const|mut)?\s*/;
const DYN_PREFIX_RE = /^dyn\s+/;
const ENUM_VARIANT_NAMES = new Set(['Some', 'None', 'Ok', 'Err']);
// `dyn Trait`, `&dyn Trait`, `Box<dyn Trait>` all name a trait object whose
// receiver-dispatch target is the trait itself (#2604) — strip the `dyn`
// keyword and any auto-trait/lifetime bound list (`dyn Trait + Send`) down to
// the principal trait name. Reference/pointer sigils are stripped by the
// caller first; wrapper unwrapping (Box<T> etc.) runs before this so the
// unwrapped inner text still gets the same treatment.
function stripDynBound(t: string): string {
if (!DYN_PREFIX_RE.test(t)) return t;
t = t.replace(DYN_PREFIX_RE, '');
const plus = t.indexOf('+');
if (plus !== -1) t = t.slice(0, plus);
return t.trim();
}
// ─── interpretImport ──────────────────────────────────────────────────────
export function interpretRustImport(captures: CaptureMatch): ParsedImport | null {
@@ -113,7 +98,6 @@ export function normalizeRustTypeName(text: string): string {
const inner = extractFirstGenericArg(t);
if (inner !== null) t = inner;
}
t = stripDynBound(t);
const bracket = t.indexOf('<');
if (bracket !== -1) t = t.slice(0, bracket);
// Take last segment of qualified paths (crate::foo::Bar → Bar)
@@ -174,7 +158,6 @@ function normalizeRustReturnType(text: string): string {
}
}
}
t = stripDynBound(t);
const bracket = t.indexOf('<');
if (bracket !== -1) t = t.slice(0, bracket);
const lastColon = t.lastIndexOf('::');
@@ -10,7 +10,6 @@ const RUST_SCOPE_QUERY = `
(enum_item) @scope.class
(union_item) @scope.class
(function_item) @scope.function
(function_signature_item) @scope.function
(closure_expression) @scope.function
(block) @scope.block
(if_expression) @scope.block
@@ -56,14 +55,6 @@ const RUST_SCOPE_QUERY = `
(function_item
name: (identifier) @declaration.name) @declaration.function
;; Declarations — trait method signature (required method, no body,
;; e.g. fn foo(self) -> T; inside a trait body). Without this, an abstract
;; trait method is invisible to scope resolution — never owned by its
;; trait's Class scope, so a dyn Trait receiver can never dispatch to
;; it (#2604).
(function_signature_item
name: (identifier) @declaration.name) @declaration.function
;; Declarations — struct fields
(field_declaration
name: (field_identifier) @declaration.name
@@ -20,7 +20,6 @@ export {
scopeResolutionPhase,
type ScopeResolutionOutput,
} from '../scope-resolution/pipeline/phase.js';
export { springConfigPhase, type SpringConfigOutput } from './spring-config.js';
export { pruneLocalSymbolsPhase, type PruneLocalSymbolsOutput } from './prune-local-symbols.js';
export { taintSummariesPhase, type TaintSummariesOutput } from './taint-summaries.js';
export { callSummariesPhase, type CallSummariesOutput } from './call-summaries.js';
@@ -25,17 +25,6 @@ export interface ProcessesOutput {
processResult: ProcessDetectionResult;
}
/**
* Compute the dynamic max-processes budget from the symbol count.
*
* Scales proportionally (symbolCount / 10) with a floor of 20.
* Prior to #2198 this was capped at 300 via `Math.min(300, …)`,
* silently truncating process detection on large repositories.
*/
export function computeDynamicMaxProcesses(symbolCount: number): number {
return Math.max(20, Math.round(symbolCount / 10));
}
export const processesPhase: PipelinePhase<ProcessesOutput> = {
name: 'processes',
// `structure` supplies `totalFiles` (progress counter) without the spurious
@@ -64,7 +53,7 @@ export const processesPhase: PipelinePhase<ProcessesOutput> = {
ctx.graph.forEachNode((n) => {
if (n.label !== 'File') symbolCount++;
});
const dynamicMaxProcesses = computeDynamicMaxProcesses(symbolCount);
const dynamicMaxProcesses = Math.max(20, Math.min(300, Math.round(symbolCount / 10)));
const processResult = await processProcesses(
ctx.graph,
@@ -1,551 +0,0 @@
/**
* Phase: springConfig
*
* Adds key-only nodes for statically readable Spring
* `application*.properties` / `application*.yml` / `application*.yaml` files.
* Language-specific ScopeResolver hooks attach consumers later. Configuration
* values are deliberately never copied into the graph because they may contain
* credentials and key identity is sufficient for impact analysis.
*
* @deps structure
* @reads Spring application configuration files
* @writes Property nodes and DEFINES edges
*/
import fs from 'node:fs/promises';
import path from 'node:path';
import { createRequire } from 'node:module';
import type { Event as YamlEvent } from 'js-yaml';
import { SPRING_CONFIG_DESCRIPTION } from '../frameworks/spring/config-bindings.js';
import { generateId } from '../../../lib/utils.js';
import type { PipelineContext, PipelinePhase, PhaseResult } from './types.js';
import { getPhaseOutput } from './types.js';
import type { StructureOutput } from './structure.js';
const require = createRequire(import.meta.url);
const yaml = require('js-yaml') as typeof import('js-yaml');
// js-yaml 5 dropped DEFAULT_SCHEMA; CORE plus these tags is what it used to be, so
// explicitly tagged values keep parsing instead of throwing (an unknown tag aborts
// the whole file). None of them can execute code.
const SPRING_YAML_SCHEMA = yaml.CORE_SCHEMA.withTags(
yaml.mergeTag,
yaml.timestampTag,
yaml.binaryTag,
yaml.omapTag,
yaml.pairsTag,
yaml.setTag,
);
const MAX_CONFIG_FILE_BYTES = 2 * 1024 * 1024;
const MAX_YAML_TRAVERSAL_DEPTH = 128;
const MAX_YAML_TRAVERSAL_NODES = 100_000;
export interface SpringConfigKey {
readonly key: string;
readonly filePath: string;
readonly line: number;
readonly profile?: string;
readonly format: 'properties' | 'yaml';
}
interface SpringConfigFile {
readonly filePath: string;
readonly profile?: string;
readonly format: SpringConfigKey['format'];
}
export interface SpringConfigOutput {
readonly configKeys: number;
}
/** Match only Spring Boot's conventional application config file names. */
export function classifySpringConfigFile(filePath: string): SpringConfigFile | null {
const base = path.posix.basename(filePath.replaceAll('\\', '/'));
const match = /^application(?:-([^.]+))?\.(properties|ya?ml)$/i.exec(base);
if (match === null) return null;
return {
filePath,
...(match[1] ? { profile: match[1] } : {}),
format: match[2].toLowerCase() === 'properties' ? 'properties' : 'yaml',
};
}
function unescapePropertyKey(raw: string): string {
return raw
.replace(/\\u([0-9a-fA-F]{4})/g, (_match, hex: string) =>
String.fromCharCode(Number.parseInt(hex, 16)),
)
.replace(/\\([:=#!\\ ])/g, '$1');
}
function logicalPropertiesLines(content: string): Array<{ text: string; line: number }> {
const physical = content.split(/\r?\n/);
const logical: Array<{ text: string; line: number }> = [];
let current = '';
let startLine = 1;
for (let index = 0; index < physical.length; index++) {
const line = physical[index];
if (current.length === 0) startLine = index + 1;
current += current.length === 0 ? line : line.trimStart();
let trailingBackslashes = 0;
for (let cursor = current.length - 1; cursor >= 0 && current[cursor] === '\\'; cursor--) {
trailingBackslashes++;
}
if (trailingBackslashes % 2 === 1) {
current = current.slice(0, -1);
continue;
}
logical.push({ text: current, line: startLine });
current = '';
}
if (current.length > 0) logical.push({ text: current, line: startLine });
return logical;
}
/** Parse `.properties` keys without retaining their values. */
export function parseSpringProperties(
content: string,
filePath: string,
profile?: string,
): SpringConfigKey[] {
const keys: SpringConfigKey[] = [];
const seen = new Set<string>();
for (const logical of logicalPropertiesLines(content)) {
const trimmed = logical.text.trimStart();
if (trimmed.length === 0 || trimmed.startsWith('#') || trimmed.startsWith('!')) continue;
let separator = -1;
let escaped = false;
for (let index = 0; index < trimmed.length; index++) {
const char = trimmed[index];
if (!escaped && (char === '=' || char === ':' || /\s/.test(char))) {
separator = index;
break;
}
escaped = !escaped && char === '\\';
if (char !== '\\') escaped = false;
}
const rawKey = (separator === -1 ? trimmed : trimmed.slice(0, separator)).trim();
const key = unescapePropertyKey(rawKey);
if (key.length === 0 || seen.has(key)) continue;
seen.add(key);
keys.push({
key,
filePath,
line: logical.line,
...(profile ? { profile } : {}),
format: 'properties',
});
}
return keys;
}
interface YamlParseEvent {
readonly startLine: number;
readonly kind: 'scalar' | 'sequence' | 'mapping' | 'alias' | null;
readonly result: unknown;
readonly aliasOf: YamlParseEvent | undefined;
readonly children: YamlParseEvent[];
}
interface YamlMappingLocation {
readonly valueEvent: YamlParseEvent;
readonly line: number;
}
interface YamlTraversalState {
remainingNodes: number;
readonly activeObjects: Set<object>;
}
function consumeYamlTraversalBudget(state: YamlTraversalState, depth: number): void {
if (depth > MAX_YAML_TRAVERSAL_DEPTH) {
throw new Error(`Spring YAML traversal depth exceeds ${MAX_YAML_TRAVERSAL_DEPTH}`);
}
state.remainingNodes--;
if (state.remainingNodes < 0) {
throw new Error(`Spring YAML traversal exceeds ${MAX_YAML_TRAVERSAL_NODES} nodes`);
}
}
function isObjectValue(value: unknown): value is object {
return value !== null && typeof value === 'object';
}
// Aliases are resolved to their anchor event by name while the tree is built,
// so following one here is a single pointer hop.
function resolveYamlAliasEvent(event: YamlParseEvent | undefined): YamlParseEvent | undefined {
return event?.aliasOf ?? event;
}
function yamlMappingPairs(event: YamlParseEvent): Array<{
key: string;
keyEvent: YamlParseEvent;
valueEvent: YamlParseEvent;
}> {
const pairs: Array<{ key: string; keyEvent: YamlParseEvent; valueEvent: YamlParseEvent }> = [];
for (let index = 0; index + 1 < event.children.length; index += 2) {
const keyEvent = event.children[index];
const valueEvent = event.children[index + 1];
if (keyEvent.kind !== 'scalar') continue;
pairs.push({ key: String(keyEvent.result), keyEvent, valueEvent });
}
return pairs;
}
/**
* First match in a pre-order walk of `event`, following sequences and `<<` merge
* chains. Iterative: children are pushed in reverse so the explicit stack pops
* them in declaration order, which is what makes "first match" mean the same
* thing it did when this recursed.
*/
function findYamlMappingLocation(
event: YamlParseEvent | undefined,
key: string,
traversal: YamlTraversalState,
): YamlMappingLocation | undefined {
const visited = new Set<YamlParseEvent>();
const stack: Array<{ event: YamlParseEvent | undefined; depth: number }> = [{ event, depth: 0 }];
while (stack.length > 0) {
const step = stack.pop();
if (step === undefined) break;
consumeYamlTraversalBudget(traversal, step.depth);
const resolved = resolveYamlAliasEvent(step.event);
if (resolved === undefined || visited.has(resolved)) continue;
visited.add(resolved);
if (resolved.kind === 'sequence') {
for (let index = resolved.children.length - 1; index >= 0; index--) {
stack.push({ event: resolved.children[index], depth: step.depth + 1 });
}
continue;
}
if (resolved.kind !== 'mapping') continue;
const pairs = yamlMappingPairs(resolved);
const direct = pairs.find((pair) => pair.key === key);
if (direct !== undefined) {
return { valueEvent: direct.valueEvent, line: direct.keyEvent.startLine };
}
const merges = pairs.filter((pair) => pair.key === '<<');
for (let index = merges.length - 1; index >= 0; index--) {
stack.push({ event: merges[index].valueEvent, depth: step.depth + 1 });
}
}
return undefined;
}
type YamlFlattenStep =
| {
readonly kind: 'visit';
readonly value: unknown;
readonly event: YamlParseEvent | undefined;
readonly prefix: string;
readonly sourceLine: number;
readonly depth: number;
}
// Pops after every descendant of the object that pushed it, which is where the
// recursive form's `finally` used to release the cycle guard.
| { readonly kind: 'leave'; readonly object: object };
/**
* Flatten a document to `dotted.key -> line`, iteratively. Children are pushed in
* reverse so the stack pops them in declaration order, keeping `out` in the same
* insertion order — and the traversal budget consumed in the same sequence — as
* the recursive walk this replaced.
*/
function flattenYamlValue(
value: unknown,
event: YamlParseEvent | undefined,
prefix: string,
out: Map<string, number>,
traversal: YamlTraversalState,
): void {
const stack: YamlFlattenStep[] = [
{ kind: 'visit', value, event, prefix, sourceLine: event?.startLine ?? 1, depth: 0 },
];
while (stack.length > 0) {
const step = stack.pop();
if (step === undefined) break;
if (step.kind === 'leave') {
traversal.activeObjects.delete(step.object);
continue;
}
const { value: current, prefix: currentPrefix, sourceLine, depth } = step;
consumeYamlTraversalBudget(traversal, depth);
const resolvedEvent = resolveYamlAliasEvent(step.event);
const trackedObject = isObjectValue(current) ? current : undefined;
if (trackedObject !== undefined) {
if (traversal.activeObjects.has(trackedObject)) continue;
traversal.activeObjects.add(trackedObject);
stack.push({ kind: 'leave', object: trackedObject });
}
if (Array.isArray(current)) {
if (current.length === 0 && currentPrefix.length > 0 && !out.has(currentPrefix)) {
out.set(currentPrefix, sourceLine);
}
for (let index = current.length - 1; index >= 0; index--) {
stack.push({
kind: 'visit',
value: current[index],
event: resolvedEvent?.children[index],
prefix: `${currentPrefix}[${index}]`,
sourceLine,
depth: depth + 1,
});
}
continue;
}
if (
current !== null &&
typeof current === 'object' &&
(resolvedEvent?.kind === 'mapping' || resolvedEvent === undefined)
) {
// js-yaml 5 builds `!!set` as a native Set, whose members are not own
// properties; v4 built a plain `{member: null}` object. Enumerate them so a
// tagged set still contributes one key per member instead of a bare leaf.
const entries: Array<[string, unknown]> =
current instanceof Set
? [...current].map((member) => [String(member), null])
: Object.entries(current as Record<string, unknown>);
if (entries.length === 0 && currentPrefix.length > 0 && !out.has(currentPrefix)) {
out.set(currentPrefix, sourceLine);
}
for (let index = entries.length - 1; index >= 0; index--) {
const [key, nested] = entries[index];
const location = findYamlMappingLocation(resolvedEvent, key, traversal);
stack.push({
kind: 'visit',
value: nested,
event: location?.valueEvent,
prefix: currentPrefix.length === 0 ? key : `${currentPrefix}.${key}`,
sourceLine: location?.line ?? sourceLine,
depth: depth + 1,
});
}
continue;
}
if (currentPrefix.length > 0 && !out.has(currentPrefix)) out.set(currentPrefix, sourceLine);
}
}
// js-yaml 5 reports node positions as source offsets; map them to 1-based lines.
function makeLineResolver(source: string): (offset: number) => number {
const lineStarts = [0];
for (let index = 0; index < source.length; index++) {
if (source[index] === '\n') lineStarts.push(index + 1);
}
return (offset: number): number => {
let low = 0;
let high = lineStarts.length - 1;
let line = 0;
while (low <= high) {
const mid = (low + high) >> 1;
if (lineStarts[mid] <= offset) {
line = mid;
low = mid + 1;
} else {
high = mid - 1;
}
}
return line + 1;
};
}
/**
* Rebuild the parse tree from js-yaml 5's event stream (v4's `listener` option
* was removed). Returns each document's root event, with aliases already
* resolved to their anchor event so merged/aliased keys keep the line where
* they were declared.
*
* One node per event, so this pass is bounded by MAX_CONFIG_FILE_BYTES alone —
* MAX_YAML_TRAVERSAL_NODES governs the later walk, which can revisit a shared
* anchor many times and so needs a budget this linear pass does not.
*/
function buildYamlEventTree(
events: readonly YamlEvent[],
source: string,
): Array<YamlParseEvent | undefined> {
const lineOf = makeLineResolver(source);
const anchors = new Map<string, YamlParseEvent>();
const stack: YamlParseEvent[] = [];
const documentRoots: Array<YamlParseEvent | undefined> = [];
const anchorName = (start: number, end: number): string | null =>
start >= 0 && end > start ? source.slice(start, end) : null;
const attach = (node: YamlParseEvent): void => {
stack[stack.length - 1]?.children.push(node);
};
const register = (name: string | null, node: YamlParseEvent): void => {
if (name !== null) anchors.set(name, node);
};
for (const event of events) {
switch (event.type) {
case yaml.EVENT_DOCUMENT:
// Anchors are document-scoped. constructFromEvents already rejects a
// cross-document alias before we get here, so this only keeps the two
// layers from disagreeing.
anchors.clear();
stack.push({
startLine: 1,
kind: null,
result: undefined,
aliasOf: undefined,
children: [],
});
break;
case yaml.EVENT_MAPPING:
case yaml.EVENT_SEQUENCE: {
const node: YamlParseEvent = {
startLine: lineOf(event.start),
kind: event.type === yaml.EVENT_MAPPING ? 'mapping' : 'sequence',
result: undefined,
aliasOf: undefined,
children: [],
};
register(anchorName(event.anchorStart, event.anchorEnd), node);
attach(node);
stack.push(node);
break;
}
case yaml.EVENT_SCALAR: {
const node: YamlParseEvent = {
startLine: lineOf(event.valueStart),
kind: 'scalar',
result: yaml.getScalarValue(source, event),
aliasOf: undefined,
children: [],
};
register(anchorName(event.anchorStart, event.anchorEnd), node);
attach(node);
break;
}
case yaml.EVENT_ALIAS: {
const target = anchors.get(anchorName(event.anchorStart, event.anchorEnd) ?? '');
attach({ startLine: 1, kind: 'alias', result: undefined, aliasOf: target, children: [] });
break;
}
case yaml.EVENT_POP: {
const done = stack.pop();
// Documents are the only top-level containers, so a pop that empties the
// stack closes a document; its single child is the document's root value.
if (done !== undefined && stack.length === 0) documentRoots.push(done.children[0]);
break;
}
}
}
return documentRoots;
}
/** Parse and flatten YAML leaves without retaining their values. */
export function parseSpringYaml(
content: string,
filePath: string,
profile?: string,
): SpringConfigKey[] {
const flattened = new Map<string, number>();
const traversal: YamlTraversalState = {
remainingNodes: MAX_YAML_TRAVERSAL_NODES,
activeObjects: new Set<object>(),
};
const events = yaml.parseEvents(content, { maxDepth: MAX_YAML_TRAVERSAL_DEPTH });
const documents = yaml.constructFromEvents(events, {
source: content,
schema: SPRING_YAML_SCHEMA,
json: true,
});
const documentEvents = buildYamlEventTree(events, content);
documents.forEach((document, index) =>
flattenYamlValue(document, documentEvents[index], '', flattened, traversal),
);
return [...flattened.entries()]
.sort(([left], [right]) => left.localeCompare(right))
.map(([key, line]) => ({
key,
filePath,
line,
...(profile ? { profile } : {}),
format: 'yaml' as const,
}));
}
function configKeyNodeId(entry: SpringConfigKey): string {
return generateId('Property', `spring-config:${entry.filePath}:${entry.key}`);
}
async function readConfigKeys(
repoPath: string,
scannedFiles: StructureOutput['scannedFiles'],
): Promise<SpringConfigKey[]> {
const keys: SpringConfigKey[] = [];
for (const scanned of scannedFiles) {
const classified = classifySpringConfigFile(scanned.path);
if (classified === null || scanned.size > MAX_CONFIG_FILE_BYTES) continue;
try {
const content = await fs.readFile(path.join(repoPath, scanned.path), 'utf8');
keys.push(
...(classified.format === 'properties'
? parseSpringProperties(content, classified.filePath, classified.profile)
: parseSpringYaml(content, classified.filePath, classified.profile)),
);
} catch {
// Malformed configuration is not a reason to fail the entire code index.
// Fail closed: no keys and therefore no misleading bindings for this file.
}
}
return keys;
}
export const springConfigPhase: PipelinePhase<SpringConfigOutput> = {
name: 'springConfig',
deps: ['structure'],
async execute(
ctx: PipelineContext,
deps: ReadonlyMap<string, PhaseResult<unknown>>,
): Promise<SpringConfigOutput> {
const { scannedFiles } = getPhaseOutput<StructureOutput>(deps, 'structure');
const configKeys = await readConfigKeys(ctx.repoPath, scannedFiles);
for (const entry of configKeys) {
const nodeId = configKeyNodeId(entry);
ctx.graph.addNode({
id: nodeId,
label: 'Property',
properties: {
name: entry.key,
filePath: entry.filePath,
startLine: entry.line,
endLine: entry.line,
description: entry.profile
? `${SPRING_CONFIG_DESCRIPTION} (profile: ${entry.profile})`
: SPRING_CONFIG_DESCRIPTION,
},
});
const fileId = generateId('File', entry.filePath);
if (ctx.graph.getNode(fileId) !== undefined) {
ctx.graph.addRelationship({
id: generateId('DEFINES', `${fileId}->${nodeId}`),
sourceId: fileId,
targetId: nodeId,
type: 'DEFINES',
confidence: 1,
reason: 'spring-config:key',
});
}
}
return { configKeys: configKeys.length };
},
};
+1 -3
View File
@@ -31,7 +31,6 @@ import {
ormPhase,
crossFilePhase,
scopeResolutionPhase,
springConfigPhase,
pruneLocalSymbolsPhase,
taintSummariesPhase,
callSummariesPhase,
@@ -243,7 +242,7 @@ export interface PipelineOptions {
*
* Phase dependency graph:
*
* scan → structure → [springConfig, markdown, cobol] → parse → [routes, tools, orm]
* scan → structure → [markdown, cobol] → parse → [routes, tools, orm]
* → crossFile → scopeResolution → pruneLocalSymbols
* → mro → di → communities → processes
*
@@ -262,7 +261,6 @@ export function buildPhaseList(options?: PipelineOptions): PipelinePhase[] {
new PhaseRegistry<PipelineOptions>()
.register(scanPhase)
.register(structurePhase)
.register(springConfigPhase)
.register(markdownPhase)
.register(cobolPhase)
.register(parsePhase)
@@ -982,15 +982,13 @@ function followChainedRef(start: TypeRef, draftById: ReadonlyMap<ScopeId, ScopeD
* name in the same scope. Higher number wins; ties keep the later match
* (last-write-wins preserves historical order within a tier).
*
* Rationale: explicit variable and field annotations always beat bindings
* derived from parameter annotations or inference because they reflect the
* most specific user intent. `self`/`cls` are treated as strongly as other
* declared types because they are language-required receiver types.
* Rationale: explicit annotations always beat inferred ones because they
* reflect user intent. `self`/`cls` are treated as strongly as annotations
* because they are language-required receiver types.
*/
function typeBindingStrength(source: TypeRef['source']): number {
switch (source) {
case 'annotation':
return 3;
case 'parameter-annotation':
case 'return-annotation':
case 'self':
@@ -743,11 +743,10 @@ export const PYTHON_QUERIES = `
// Java queries - works with tree-sitter-java
export const JAVA_QUERIES = `
; Classes, Interfaces, Enums, Records, Annotations
; Classes, Interfaces, Enums, Annotations
(class_declaration name: (identifier) @name) @definition.class
(interface_declaration name: (identifier) @name) @definition.interface
(enum_declaration name: (identifier) @name) @definition.enum
(record_declaration name: (identifier) @name) @definition.record
(annotation_type_declaration name: (identifier) @name) @definition.annotation
; Anonymous class bodies: new Runnable() { ... } — no @name capture; the
@@ -205,19 +205,6 @@ export interface WorkerPoolOptions {
* created. Default `Math.max(3, poolSize)`.
*/
consecutiveFailureThreshold?: number;
/**
* Startup budget in milliseconds for a replacement worker to emit the
* `{type:'ready'}` handshake before the pool treats it as a startup
* crash (see {@link waitForWorkerReady}). Default 5000; also overridable
* via `GITNEXUS_WORKER_READY_TIMEOUT_MS`, mirroring
* `GITNEXUS_WORKER_SUB_BATCH_TIMEOUT_MS`. On a slow or heavily loaded
* host, a full pool of workers cold-starting concurrently can
* legitimately need more than 5s to load the native grammar bindings —
* without the override every slot times out and the pool misclassifies
* the slow start as a deterministic startup crash-loop, aborting the
* whole analyze.
*/
workerReadyTimeoutMs?: number;
/**
* Test-only injection point for the Worker constructor. When provided,
* the pool uses this factory instead of `new Worker(workerUrl)`. Production
@@ -419,7 +406,17 @@ const DEFAULT_TIMEOUT_BACKOFF_FACTOR = 2;
const DEFAULT_MAX_RESPAWNS_PER_SLOT = 3;
const DEFAULT_MAX_CUMULATIVE_TIMEOUT_FACTOR = 5;
const DEFAULT_CONSECUTIVE_FAILURE_THRESHOLD_FLOOR = 3;
const DEFAULT_WORKER_READY_TIMEOUT_MS = 5_000;
/**
* Bounded wait for a replacement worker to emit the `{type:'ready'}`
* handshake from `parse-worker.ts`. Trusting Node's `online` event alone
* lets a worker that crashes during top-of-script init slip past pool
* startup — the pool only notices on the first dispatch's idle timeout
* (default 30s). 5 seconds is a generous budget for parser + grammar
* imports; if the worker hasn't reported ready by then, it's almost
* certainly stuck or crashed and the pool should surface the failure
* fast rather than wait out the dispatch idle timeout.
*/
const WORKER_READY_TIMEOUT_MS = 5_000;
/**
* Default upper bound on auto-resolved pool size. Past 16 workers the
* dominant cost shifts from worker-side parsing to main-thread merge /
@@ -550,7 +547,6 @@ interface ResolvedWorkerPoolOptions {
maxCumulativeTimeoutMs: number;
consecutiveFailureThreshold: number;
shutdownDrainMs: number;
workerReadyTimeoutMs: number;
}
export function resolveWorkerPoolOptions(
@@ -587,10 +583,6 @@ export function resolveWorkerPoolOptions(
nonNegativeInteger(options.shutdownDrainMs) ??
nonNegativeInteger(process.env.GITNEXUS_WORKER_SHUTDOWN_DRAIN_MS) ??
DEFAULT_SHUTDOWN_DRAIN_MS,
workerReadyTimeoutMs:
positiveInteger(options.workerReadyTimeoutMs) ??
positiveInteger(process.env.GITNEXUS_WORKER_READY_TIMEOUT_MS) ??
DEFAULT_WORKER_READY_TIMEOUT_MS,
};
}
@@ -691,27 +683,6 @@ function captureWorkerStderr(worker: Worker): void {
stream.on('error', () => undefined);
}
/**
* Forward a worker's piped stdout to the parent process's stdout, so worker
* logs stay visible now that the production factory spawns with
* `{ stdout: true }`. Workers with INHERITED stdout have been observed to
* crash silently during top-of-script init (exit code 1, nothing on stderr,
* roughly half of a concurrently spawned pool) on macOS 26.5 under both
* Node 22 and 26; piping stdout eliminates the crash entirely. Piping also
* matches the existing stderr handling, so worker output no longer races the
* parent's raw fd. No-op when the worker has no `stdout` stream (test
* factories).
*/
function forwardWorkerStdout(worker: Worker): void {
const stream = worker.stdout;
if (!stream) return;
stream.on('data', (chunk: Buffer | string) => {
process.stdout.write(chunk);
});
// A stdout stream error must never crash the pool.
stream.on('error', () => undefined);
}
/** Captured stderr tail for a worker, trimmed; '' when nothing was captured. */
function workerStderrTail(worker: Worker): string {
return workerStderrTails.get(worker)?.text.trim() ?? '';
@@ -751,14 +722,13 @@ function workerErrorReason(workerIndex: number, message: string, stack?: string)
* (parser/grammar import failure, missing native binding) slip past
* pool startup. The pool then only noticed the dead replacement on the
* first dispatch's idle timeout (default 30s) — a long stall masking
* an actual crash. This handshake bounds the wait at `readyTimeoutMs`
* (see {@link WorkerPoolOptions.workerReadyTimeoutMs}) and surfaces init
* failures as `error` / `exit` / `messageerror` events directly.
* `messageerror` is wired the same way: a V8 deserialization failure
* during startup is treated as worker death and rejects the readiness
* promise.
* an actual crash. This handshake bounds the wait at
* {@link WORKER_READY_TIMEOUT_MS} and surfaces init failures as
* `error` / `exit` / `messageerror` events directly. `messageerror` is
* wired the same way: a V8 deserialization failure during startup is
* treated as worker death and rejects the readiness promise.
*/
function waitForWorkerReady(worker: Worker, readyTimeoutMs: number): Promise<void> {
function waitForWorkerReady(worker: Worker): Promise<void> {
return new Promise<void>((resolve, reject) => {
const cleanup = () => {
clearTimeout(timer);
@@ -811,11 +781,11 @@ function waitForWorkerReady(worker: Worker, readyTimeoutMs: number): Promise<voi
new Error(
withStderr(
worker,
`Replacement worker did not report ready within ${readyTimeoutMs}ms — likely crashed during top-of-script init (slow host? raise GITNEXUS_WORKER_READY_TIMEOUT_MS)`,
`Replacement worker did not report ready within ${WORKER_READY_TIMEOUT_MS}ms — likely crashed during top-of-script init`,
),
),
);
}, readyTimeoutMs);
}, WORKER_READY_TIMEOUT_MS);
worker.on('message', onMessage);
worker.once('error', onError);
worker.once('exit', onExit);
@@ -961,10 +931,6 @@ export const createWorkerPool = (
options?.workerFactory ??
((url: URL) =>
new Worker(url, {
// Piped (not inherited) stdio: stderr for crash capture (#1741),
// stdout because inherited stdout triggers silent startup crashes on
// some hosts (see forwardWorkerStdout).
stdout: true,
stderr: true,
workerData: workerStoreData,
// The CFG visitors build per-function control-flow graphs by RECURSIVE
@@ -978,11 +944,10 @@ export const createWorkerPool = (
// try/catch) and only that function's PDG is skipped, never a crash.
resourceLimits: { stackSizeMb: 16 },
}));
/** Spawn + wire stdio capture/forwarding in one step (used by all spawn sites). */
/** Spawn + wire stderr capture in one step (used by all spawn sites). */
const spawnAndCapture = (url: URL): Worker => {
const worker = spawnWorker(url);
captureWorkerStderr(worker);
forwardWorkerStdout(worker);
return worker;
};
const workers: (Worker | undefined)[] = new Array(size);
@@ -1134,7 +1099,7 @@ export const createWorkerPool = (
const worker = workers[i];
if (!worker) return; // terminated mid-startup
try {
await waitForWorkerReady(worker, poolOptions.workerReadyTimeoutMs);
await waitForWorkerReady(worker);
anyWorkerReachedReady = true;
return; // ready — slot stays in activeSlots
} catch (err) {
@@ -1196,7 +1161,7 @@ export const createWorkerPool = (
chunkHash?: string,
): Promise<TResult[]> => {
// Await the initial-spawn readiness gate (F13). On first dispatch
// this blocks for up to poolOptions.workerReadyTimeoutMs while every initial
// this blocks for up to WORKER_READY_TIMEOUT_MS while every initial
// worker's `{type:'ready'}` handshake is checked; on subsequent
// dispatches the promise is already settled and resolves
// synchronously. Slots whose initial worker crashed in top-of-
@@ -1395,7 +1360,7 @@ export const createWorkerPool = (
if (stopped) return false;
const replacement = spawnAndCapture(workerUrl);
try {
await waitForWorkerReady(replacement, poolOptions.workerReadyTimeoutMs);
await waitForWorkerReady(replacement);
} catch (err) {
await replacement.terminate().catch(() => undefined);
logger.warn(
+31 -43
View File
@@ -101,24 +101,18 @@ const POSIX_MISSING_DEPENDENCY_SIGNATURES: readonly RegExp[] = [
* display language — the only localized part is the OS-error tail after it. So
* it is the language-independent fallback signal once the specific tails miss: a
* French/German/Japanese Windows 126 has a localized tail we cannot enumerate,
* but it still carries this wrapper. See hedgedLoadFailureRemedy.
* but it still carries this wrapper. See HEDGED_LOAD_FAILURE_REMEDY.
*/
const LOAD_FAILURE_WRAPPER = /failed to load library/i;
// Remedies are label-parameterized (#2623 follow-up): doctor now live-probes
// VECTOR through the same classifier, and FTS-specific advice (`--repair-fts`
// repairs FTS indexes only) must not be dispensed for other extensions.
const repairFtsHint = (label: string, lead: string): string =>
label === 'FTS' ? ` (${lead}\`gitnexus analyze --repair-fts\`)` : '';
const MISSING_FILE_REMEDY =
'The FTS extension is not installed. Re-run with network access and ' +
'GITNEXUS_LBUG_EXTENSION_INSTALL=auto (or `gitnexus analyze --repair-fts`) to download it.';
const missingFileRemedy = (label: string): string =>
`The ${label} extension is not installed. Re-run with network access and ` +
`GITNEXUS_LBUG_EXTENSION_INSTALL=auto${repairFtsHint(label, 'or ')} to download it.`;
const corruptFileRemedy = (label: string): string =>
`The ${label} extension file is present but unreadable (corrupt, truncated, or built for another ` +
`platform). Re-download it with network access and ` +
`GITNEXUS_LBUG_EXTENSION_INSTALL=auto${repairFtsHint(label, '')}.`;
const CORRUPT_FILE_REMEDY =
'The FTS extension file is present but unreadable (corrupt, truncated, or built for another ' +
'platform). Re-download it with network access and GITNEXUS_LBUG_EXTENSION_INSTALL=auto ' +
'(`gitnexus analyze --repair-fts`).';
// Single source of truth for the VC++ runtime-install pointer, shared by the
// Windows-126 and structural missing-dependency remedies so the name/URL cannot
@@ -128,15 +122,15 @@ const VC_REDIST_INSTALL_HINT =
'https://aka.ms/vs/17/release/vc_redist.x64.exe';
// MSVC-first per DuckDB's canonical answer for this exact error; OpenSSL second.
const windowsMissingDependencyRemedy = (label: string): string =>
`The ${label} extension is present but a required runtime library is missing (Windows error 126). ` +
const WINDOWS_MISSING_DEPENDENCY_REMEDY =
'The FTS extension is present but a required runtime library is missing (Windows error 126). ' +
'Reinstalling the extension will NOT help. Install ' +
VC_REDIST_INSTALL_HINT +
'; if the error persists, the extension also needs OpenSSL 3 ' +
'(libcrypto-3-x64.dll / libssl-3-x64.dll) on the DLL search path.';
const posixMissingDependencyRemedy = (label: string): string =>
`The ${label} extension is present but a shared library it depends on could not be loaded (named in ` +
const POSIX_MISSING_DEPENDENCY_REMEDY =
'The FTS extension is present but a shared library it depends on could not be loaded (named in ' +
'the error above). Reinstalling the extension will NOT help — install that library or add it to ' +
'your loader search path.';
@@ -146,18 +140,16 @@ const posixMissingDependencyRemedy = (label: string): string =>
// branches — rather than confidently prescribing the wrong single fix. The clean
// long-term fix is upstream: have LadybugDB include the numeric GetLastError/errno
// in the message (as it already does elsewhere), so this becomes a code match.
const hedgedLoadFailureRemedy = (label: string): string =>
`The ${label} extension file was found but could not be loaded — see the "Error:" text above (shown ` +
const HEDGED_LOAD_FAILURE_REMEDY =
'The FTS extension file was found but could not be loaded — see the "Error:" text above (shown ' +
"in your system's language). Reinstalling usually will not help. If it names a missing module or " +
'library, install the required runtime (on Windows: the Microsoft Visual C++ 2015-2022 ' +
'Redistributable x64 and OpenSSL 3); if it names a corrupt or invalid file, ' +
(label === 'FTS'
? 'run `gitnexus analyze --repair-fts` to re-download.'
: 're-run analyze with network access and GITNEXUS_LBUG_EXTENSION_INSTALL=auto to re-download.');
'Redistributable x64 and OpenSSL 3); if it names a corrupt or invalid file, run ' +
'`gitnexus analyze --repair-fts` to re-download.';
const unknownRemedy = (label: string): string =>
`The ${label} extension failed to load for an unrecognized reason. Run \`gitnexus doctor\` for live ` +
`${label} status and verify the extension file and platform.`;
const UNKNOWN_REMEDY =
'The FTS extension failed to load for an unrecognized reason. Run `gitnexus doctor` for live ' +
'FTS status and verify the extension file and platform.';
const matchesAny = (reason: string, signatures: readonly RegExp[]): boolean =>
signatures.some((re) => re.test(reason));
@@ -170,20 +162,19 @@ const matchesAny = (reason: string, signatures: readonly RegExp[]): boolean =>
*/
export function classifyExtensionLoadError(
reason: string | undefined | null,
label: string = 'FTS',
): ExtensionLoadDiagnosis {
const text = reason ?? '';
if (matchesAny(text, MISSING_FILE_SIGNATURES)) {
return { kind: 'missing_file', remedy: missingFileRemedy(label) };
return { kind: 'missing_file', remedy: MISSING_FILE_REMEDY };
}
if (matchesAny(text, FILE_CORRUPTION_SIGNATURES)) {
return { kind: 'corrupt_file', remedy: corruptFileRemedy(label) };
return { kind: 'corrupt_file', remedy: CORRUPT_FILE_REMEDY };
}
if (matchesAny(text, WINDOWS_MISSING_DEPENDENCY_SIGNATURES)) {
return { kind: 'missing_dependency', remedy: windowsMissingDependencyRemedy(label) };
return { kind: 'missing_dependency', remedy: WINDOWS_MISSING_DEPENDENCY_REMEDY };
}
if (matchesAny(text, POSIX_MISSING_DEPENDENCY_SIGNATURES)) {
return { kind: 'missing_dependency', remedy: posixMissingDependencyRemedy(label) };
return { kind: 'missing_dependency', remedy: POSIX_MISSING_DEPENDENCY_REMEDY };
}
// Language-independent fallback: the extension demonstrably failed to load
// (lbug's English wrapper is present) but the localized OS tail matched no
@@ -191,9 +182,9 @@ export function classifyExtensionLoadError(
// remedy — strictly better than the generic `unknown` for non-English hosts,
// and it never prescribes the wrong fix.
if (LOAD_FAILURE_WRAPPER.test(text)) {
return { kind: 'missing_dependency', remedy: hedgedLoadFailureRemedy(label) };
return { kind: 'missing_dependency', remedy: HEDGED_LOAD_FAILURE_REMEDY };
}
return { kind: 'unknown', remedy: unknownRemedy(label) };
return { kind: 'unknown', remedy: UNKNOWN_REMEDY };
}
// ── Language-independent structural layer ────────────────────────────────────
@@ -201,8 +192,8 @@ export function classifyExtensionLoadError(
/** Well-formedness of the extension binary for the host platform + arch. */
export type ExtensionBinaryState = 'absent' | 'corrupt' | 'valid' | 'indeterminate';
const structuralMissingDependencyRemedy = (label: string): string =>
`The ${label} extension file is valid, so the failure is a missing or incompatible runtime dependency, ` +
const STRUCTURAL_MISSING_DEPENDENCY_REMEDY =
'The FTS extension file is valid, so the failure is a missing or incompatible runtime dependency, ' +
'not the extension itself — reinstalling will NOT help. On Windows, install ' +
VC_REDIST_INSTALL_HINT +
' and ensure OpenSSL 3 is available; on Linux/macOS install the shared library named in the error above.';
@@ -341,16 +332,13 @@ export function inspectExtensionBinary(
* classifier (which still carries the language-independent hedged fallback). This
* is the entry point every surface should call.
*/
export function diagnoseExtensionLoad(
reason: string | undefined | null,
label: string = 'FTS',
): ExtensionLoadDiagnosis {
export function diagnoseExtensionLoad(reason: string | undefined | null): ExtensionLoadDiagnosis {
const text = reason ?? '';
const stringResult = classifyExtensionLoadError(text, label);
const stringResult = classifyExtensionLoadError(text);
const fileState = inspectExtensionBinary(extractExtensionPath(text));
if (fileState === 'corrupt') {
return { kind: 'corrupt_file', remedy: corruptFileRemedy(label) };
return { kind: 'corrupt_file', remedy: CORRUPT_FILE_REMEDY };
}
if (fileState === 'valid') {
// The structural probe only inspects the first BINARY_HEADER_BYTES, so a file
@@ -369,7 +357,7 @@ export function diagnoseExtensionLoad(
const remedy =
stringResult.kind === 'missing_dependency'
? stringResult.remedy
: structuralMissingDependencyRemedy(label);
: STRUCTURAL_MISSING_DEPENDENCY_REMEDY;
return { kind: 'missing_dependency', remedy };
}
// 'absent' or 'indeterminate' → no positive structural evidence, so defer to the
+1 -1
View File
@@ -323,7 +323,7 @@ export class ExtensionManager {
name,
loaded: false,
reason,
diagnosis: diagnoseExtensionLoad(reason, label),
diagnosis: diagnoseExtensionLoad(reason),
});
const key = `${name}:${reason}`;
if (this.warnedKeys.has(key)) return;
+14 -163
View File
@@ -23,11 +23,7 @@ import { streamAllCSVsToDisk, type StreamedCSVResult } from './csv-generator.js'
import type { PdgEmitManifest } from './pdg-emit-sink.js';
import { getNodeLabel as deriveNodeLabel, type WriteStreamFactory } from './rel-pair-routing.js';
import { EMBEDDABLE_LABELS, type CachedEmbedding } from '../embeddings/types.js';
import {
extensionManager,
resolveAnalyzeInstallPolicy,
type ExtensionEnsureOptions,
} from './extension-loader.js';
import { extensionManager, type ExtensionEnsureOptions } from './extension-loader.js';
import {
classifyDeleteAllError,
closeLbugConnection,
@@ -36,7 +32,6 @@ import {
isDbBusyError,
isOpenRetryExhausted,
isWalCorruptionError,
bufferPoolExhaustionRemedy,
openLbugConnection,
sleep,
toNativeSafePath,
@@ -46,7 +41,6 @@ import {
type LbugConnectionHandle,
} from './lbug-config.js';
import {
cleanQuarantinedMissingShadowWals,
finalizeLbugSidecarsAfterClose,
guardWalQuarantine,
isMissingShadowSidecarError,
@@ -56,8 +50,8 @@ import {
quarantineWalForMissingShadow,
renameFailureMessage,
shadowSidecarRecoveryMessage,
sidecarPreflightDisabled,
} from './sidecar-recovery.js';
import { isVectorExtensionSupportedByPlatform } from '../platform/capabilities.js';
import { logger } from '../logger.js';
// ---------------------------------------------------------------------------
@@ -824,30 +818,6 @@ const doInitLbug = async (dbPath: string, readOnly: boolean = false) => {
// -------------------------------------------------------------------------
const releaseInitLock = await acquireInitLock(dbPath);
try {
// Reclaim missing-shadow WAL quarantines from a PRIOR crash (#2637).
// LadybugDB renames an unrecoverable WAL aside as
// `${dbPath}.wal.missing-shadow.<ts>-<rand>` (quarantineWalForMissingShadow)
// instead of deleting it. Once quarantined it is permanently detached from
// the live store and never reopened, so reclaiming it is safe regardless of
// whether the main DB file exists this run — unlike the orphan-sidecar
// cleanup below, this must NOT be gated on "main DB missing": a quarantine
// event and a healthy main DB are independent facts. Never let a reclaim
// failure (e.g. a transient EBUSY from an antivirus scan) block DB startup.
if (!sidecarPreflightDisabled()) {
try {
const reclaimed = await cleanQuarantinedMissingShadowWals(dbPath);
for (const file of reclaimed) {
logger.warn(
`GitNexus: reclaimed quarantined WAL ${path.basename(file)} from a prior crash`,
);
}
} catch (err) {
logger.warn(
`GitNexus: failed to reclaim missing-shadow WAL quarantines: ${summarizeError(err)}`,
);
}
}
// Crash-recovery cleanup: if the main DB file is missing, stale sidecars
// from an interrupted run can block fresh opens indefinitely.
try {
@@ -979,14 +949,7 @@ const copyNodeCSVs = async (
const copyQuery = getCopyQuery(table, normalizeCopyPath(csvPath));
await copyCsvWithRetry(targetConn, copyQuery, (retryErr) => {
const retryMsg = retryErr instanceof Error ? retryErr.message : String(retryErr);
// Pool exhaustion gets a remedy (#2631): the raw binder text gives the
// operator nothing to act on, and on non-4K-page hosts (Ascend aarch64,
// Apple Silicon) the pool bills up to pageSize/4KiB x faster than the
// sizing was calibrated for — name the knob and the mechanism.
const remedy = bufferPoolExhaustionRemedy(retryMsg);
throw new Error(
`COPY failed for ${table}: ${retryMsg.slice(0, 200)}${remedy ? ` ${remedy}` : ''}`,
);
throw new Error(`COPY failed for ${table}: ${retryMsg.slice(0, 200)}`);
});
}
};
@@ -1158,7 +1121,6 @@ export const loadGraphToLbug = async (
const insertedRels = totalValidRels;
const warnings: string[] = [];
let poolRemedyIssued = false;
if (insertedRels > 0) {
log(`Loading edges: ${insertedRels.toLocaleString()} across ${relsByPair.size} types`);
@@ -1185,17 +1147,6 @@ export const loadGraphToLbug = async (
await copyCsvWithRetry(writeConn, copyQuery, (retryErr) => {
const retryMsg = retryErr instanceof Error ? retryErr.message : String(retryErr);
warnings.push(`${fromLabel}->${toLabel} (${rows} edges): ${retryMsg.slice(0, 80)}`);
// One remedy per bulk load, not per pair (#2631): pool exhaustion
// repeats for every remaining pair once it starts. logger.warn, not
// just warnings.push — the returned warnings array has no consumer at
// any call site, so a push alone would leave the remedy invisible
// while the row-by-row fallback quietly degrades the load.
const remedy = poolRemedyIssued ? undefined : bufferPoolExhaustionRemedy(retryMsg);
if (remedy) {
poolRemedyIssued = true;
warnings.push(remedy);
logger.warn(remedy);
}
failedPairEdges += rows;
failedPairCsvPaths.add(pairCsvPath);
});
@@ -2745,16 +2696,14 @@ export const loadVectorExtension = async (
): Promise<boolean> => {
const useModuleState = targetConn === undefined;
if (useModuleState && vectorExtensionLoaded) return true;
// No platform gate. Windows was hard-refused here for years on the strength
// of an early-era report that in-process INSTALL VECTOR could SIGSEGV
// (#1365) — but the extension server ships win_amd64 VECTOR artifacts for
// every 0.18.x extension version (probed live: v0.18.0 and v0.18.1 both
// serve a real PE32+ DLL; the pinned 0.18.2 core resolves its extension
// directory to 0.18.1, strace-verified), and INSTALL now runs in a spawned
// child process (installDuckDbExtensionOutOfProcess), so even a crashing
// installer kills only the child and degrades to `false` here. LOAD of a
// present extension file is an ordinary in-process load whose failures
// surface as catchable errors, exactly like FTS.
// INSTALL VECTOR crashes with SIGSEGV on Windows: the KuzuDB native extension
// installer has an unhandled error path on Windows that raises a fatal signal
// that JS try/catch cannot intercept. Skip loading — vector/embedding search
// is unavailable but all graph index queries still work. Do NOT set
// vectorExtensionLoaded here: the flag means "successfully loaded", and a
// subsequent call would otherwise short-circuit to `return true` at the top.
if (process.platform === 'win32') return false;
if (!isVectorExtensionSupportedByPlatform()) return false;
const c: lbug.Connection | null = targetConn ?? conn;
if (!c) {
@@ -2863,78 +2812,6 @@ export const createVectorIndex = async (): Promise<boolean> => {
}
};
/**
* Make DML against {@link EMBEDDING_TABLE_NAME} legal on the writable
* connection when it can be, and report whether it is.
*
* LadybugDB refuses EVERY mutation of a table carrying an HNSW index while
* the VECTOR extension is not loaded on that connection: `DELETE` fails with
* "Trying to delete from an index on table CodeEmbedding but its extension is
* not loaded", `CREATE` with the matching "insert into an index" variant,
* `DROP TABLE` is refused while the index references it, and `SET` — even on
* a NON-indexed property — segfaults the process outright. Probed against
* @ladybugdb/core 0.18.2 (the lockfile-pinned version) and 0.18.0 — every
* result identical on both (#2623).
*
* Dropping the index is NOT an available recovery: `CALL DROP_VECTOR_INDEX`
* is itself a VECTOR-extension function and resolves to "Catalog exception:
* function DROP_VECTOR_INDEX is not defined" in exactly the state it would
* need to rescue. Loading the extension is the only in-place repair, which is
* why this returns a verdict instead of attempting a fixup.
*
* `true` = embedding-row DML is safe: either VECTOR is now loaded, or the
* table carries no index to trip over. `false` = genuinely blocked (index
* present, extension unloadable); the analyze orchestrator answers that by
* escalating to the wipe-and-rebuild write plan instead of failing
* mid-writeback.
*
* Cheap by construction: one local `SHOW_INDEXES` read settles the common
* "this repo never built an embedding index" case without touching the
* extension machinery at all, so a VECTOR-less machine is not charged a
* bounded INSTALL attempt on every incremental analyze. `SHOW_INDEXES` is
* readable WITHOUT the extension and reports `extension_loaded` per index, so
* no error-string sniffing is needed; it runs through the unprepared
* `conn.query()` path like every other `CALL` procedure here (#2114).
*/
export const ensureEmbeddingRowDmlSafe = async (): Promise<boolean> => {
const targetConn = conn;
if (!targetConn) {
throw new Error('LadybugDB not initialized. Call initLbug first.');
}
// Catalog FIRST. The overwhelmingly common case on a repo that never enabled
// embeddings is "no index at all", and that is provable with one local read
// — no extension needed. Loading first would make every incremental analyze
// on a VECTOR-less machine pay a bounded out-of-process INSTALL attempt (the
// `auto` policy) plus an "extension unavailable" warning, for a repo that
// can never hit this hazard.
let indexRows: any[] | undefined;
try {
indexRows = await withConnLock(async () =>
readQueryRows(await targetConn.query('CALL SHOW_INDEXES() RETURN *')),
);
} catch (err) {
// Fall through to the load attempt: unable to prove the index is absent,
// so the extension is the only thing that can make DML safe.
logger.warn(
{ err },
`Could not read the index catalog to check for a ${EMBEDDING_TABLE_NAME} vector index; ` +
'falling back to loading the VECTOR extension.',
);
}
// Any non-HASH index on the embedding table gates DML. Keyed on index TYPE,
// not name, so an index built under a different name still counts; the
// implicit primary-key HASH index is engine-internal and never gates.
const indexGatesDml =
indexRows === undefined ||
indexRows.some((row) => {
const table = row?.table_name ?? row?.[0];
if (table !== EMBEDDING_TABLE_NAME) return false;
return (row?.index_type ?? row?.[2]) !== 'HASH';
});
if (!indexGatesDml) return true;
return await loadVectorExtension(undefined, { policy: resolveAnalyzeInstallPolicy() });
};
/**
* Lazy-create an FTS index, caching the fact in-process.
*
@@ -3027,30 +2904,7 @@ export const queryFTS = async (
};
/**
* True for the two benign "nothing to drop" `DROP_FTS_INDEX` failures —
* both catalog/binder exceptions, LadybugDB's classes for "this name isn't
* bound to anything right now" (probe-verified end-to-end through
* `dropFTSIndex`'s real `conn.query()` path against @ladybugdb/core
* 0.18.x): the named index was never created (`Binder exception: Table <T>
* doesn't have an index with name <name>.`), or the FTS extension/function
* isn't registered at all (`Catalog exception: function DROP_FTS_INDEX is
* not defined...`). A real engine failure — e.g. the `Runtime exception:
* FTS index '<name>' is inconsistent: ...` class from #2589 — is a
* DIFFERENT exception class (an execution-time failure, not a catalog/bind
* lookup miss), so this returns false for it. Anchored to the START of the
* message (not a bare substring search): every probed LadybugDB error leads
* with its exception class, and anchoring means a future message that merely
* mentions "Binder exception" or "Catalog exception" further in in the body
* of an otherwise-genuine failure can't be misclassified as benign. Pure
* string logic so it is unit-testable without a native LadybugDB connection.
*/
export const isBenignDropFtsIndexError = (message: string): boolean =>
message.startsWith('Binder exception:') || message.startsWith('Catalog exception:');
/**
* Drop an FTS index. Tolerates only {@link isBenignDropFtsIndexError} —
* anything else rethrows instead of being silently masked, which previously
* let a corrupted index persist across analyze runs undetected.
* Drop an FTS index
*/
export const dropFTSIndex = async (tableName: string, indexName: string): Promise<void> => {
if (!conn) {
@@ -3059,11 +2913,8 @@ export const dropFTSIndex = async (tableName: string, indexName: string): Promis
try {
await queryAndDrain(conn, `CALL DROP_FTS_INDEX('${tableName}', '${indexName}')`);
} catch (e: unknown) {
const msg = e instanceof Error ? e.message : String(e);
if (!isBenignDropFtsIndexError(msg)) {
throw e;
}
} catch {
// Index may not exist
} finally {
ensuredFTSIndexes.delete(ftsIndexKey(tableName, indexName));
}
+12 -251
View File
@@ -321,16 +321,6 @@ const resolveCheckpointThreshold = (): number => {
const DEFAULT_BUFFER_POOL_CAP = 2 * 1024 * 1024 * 1024;
const BUFFER_POOL_FLOOR = 64 * 1024 * 1024;
// COPY-safety floor for the adaptive hint (below). LadybugDB's bulk COPY needs
// working buffer-pool memory that scales with the repo: a 64 MiB pool fails
// ("buffer pool is full and no memory could be freed") on any non-trivial repo,
// and even the ~1800-file GitNexus checkout needs ≥256 MiB. So the adaptive
// size never drops a repo below this — a distinct, higher floor than
// BUFFER_POOL_FLOOR, which only guards defaultBufferPoolSize on tiny-RAM
// machines. It is still clamped up to defaultBufferPoolSize, so a machine whose
// default is below this floor keeps its default rather than over-committing.
const ADAPTIVE_POOL_FLOOR = 256 * 1024 * 1024;
const parseBufferPoolSize = (raw: string | undefined): number | undefined => {
if (raw === undefined) return undefined;
const normalized = raw.trim();
@@ -340,151 +330,19 @@ const parseBufferPoolSize = (raw: string | undefined): number | undefined => {
return Math.floor(parsed);
};
/**
* The buffer-manager frame size compiled into every shipped `@ladybugdb/core`
* binary (`LBUG_PAGE_SIZE_LOG2 = 12` in the engine's CMake) — frames are 4 KiB
* on every platform, independent of the OS page size.
*/
const LBUG_ASSUMED_FRAME_SIZE = 4096;
/**
* How much the OS page size amplifies buffer-pool consumption (#2631).
*
* LadybugDB's VM region charges pool budget per DISCARD GRANULE, not per
* frame: `discardGranuleSize = max(frameSize, osPageSize)` (vm_region.cpp),
* `claimFrame` bills the whole granule when its first 4 KiB frame becomes
* resident, and `releaseFrame` refunds only when the granule's LAST frame
* leaves. On a 64 KiB-page kernel (aarch64 openEuler — Ascend hosts) that is
* 16 frames per granule: scattered access is billed up to 16× its real bytes,
* and whole eviction passes can evict frames yet refund nothing — which is
* exactly the engine's "buffer pool is full and no memory could be freed"
* throw. Apple Silicon macOS (16 KiB pages) is the same mechanism at 4×.
*
* So the ANALYZE-path pool sizes (the per-element estimate, the COPY-safety
* floor, and the cap the hint is clamped against) are scaled by this ratio:
* the budget must cover worst-case granule charging or COPY dies on non-4K
* hosts with a pool that would be ample on x86. The hintless default
* (defaultBufferPoolSize — MCP serve, doctor, native-check) is deliberately
* NOT scaled: the pool is a native eager allocation committed at DB open
* (measured — see POOL_BYTES_PER_ELEMENT below), so scaling the global
* default would revert the #2557 OOM cap on every 16 KiB/64 KiB host. If the
* engine ever charges per-frame (or ships page-size-matched frames), this
* collapses back to 1 and the scaling disappears.
*
* Fail-safe: an undetectable page size (win32 — where the granule mechanism
* is absent anyway — or a failed `getconf`) means ratio 1, i.e. today's
* behavior.
*/
export const granuleRatio = (pageSize: number | undefined = getOsPageSize()): number => {
if (pageSize === undefined || !Number.isFinite(pageSize)) return 1;
return Math.max(1, Math.floor(pageSize / LBUG_ASSUMED_FRAME_SIZE));
};
/**
* Hintless pool default — MCP serve, doctor, native-check, any open without a
* per-run hint. Deliberately UNSCALED (#2557): the pool is an eager native
* allocation at DB open, so a page-size-scaled default would hand a
* long-lived `gitnexus mcp` on a 16 KiB/64 KiB host up to 80% of RAM — the
* exact OOM exposure the 2 GiB cap was added to remove.
*/
const defaultBufferPoolSize = (): number =>
Math.min(DEFAULT_BUFFER_POOL_CAP, Math.max(BUFFER_POOL_FLOOR, Math.floor(os.totalmem() * 0.8)));
/**
* Upper bound for the ANALYZE-path (hinted) pool: the #2557 cap scaled by the
* granule ratio, still bounded by 80% of RAM. Scaling only this bound — and
* not defaultBufferPoolSize — is what lets the #2631 fix take effect during
* the bulk COPY without touching hintless opens: with an unscaled cap the
* min() below would clamp the scaled COPY floor straight back to 2 GiB.
*/
const scaledAnalyzePoolCap = (pageSize: number | undefined): number =>
Math.min(
DEFAULT_BUFFER_POOL_CAP * granuleRatio(pageSize),
Math.max(BUFFER_POOL_FLOOR, Math.floor(os.totalmem() * 0.8)),
);
/**
* Clamp an adaptive pool request to [ADAPTIVE_POOL_FLOOR × granuleRatio,
* scaledAnalyzePoolCap]. The lower bound keeps LadybugDB's COPY viable
* (scaled because the granule accounting inflates consumption on non-4K
* hosts, see granuleRatio); the upper bound means the hint can never exceed
* the page-size-scaled #2557 cap or 80% of RAM — and on a machine whose cap
* is below the COPY floor, the cap wins, so the pool is never over-committed.
* On 4 KiB hosts (ratio 1) this is byte-identical to clamping against the
* hintless default.
*/
const clampBufferPool = (bytes: number, pageSize: number | undefined = getOsPageSize()): number =>
Math.min(
scaledAnalyzePoolCap(pageSize),
Math.max(ADAPTIVE_POOL_FLOOR * granuleRatio(pageSize), Math.floor(bytes)),
);
/**
* Buffer-pool bytes to provision per graph element (node + relationship).
*
* The fixed 2 GiB default is far larger than most repos' working set, and
* LadybugDB eagerly commits the pool at DB open — measured: a full
* `analyze --force` of the GitNexus checkout takes ~51 s with the 2 GiB pool
* vs ~35 s with the ~414 MiB this factor yields (31% faster; the oversized
* pool's commit dominates). The pool is a page cache over the on-disk index,
* which scales with node/edge count, so a per-element budget sizes it to the
* repo. Kept generous so the whole index stays resident (no COPY thrash) and
* always clamped to at least ADAPTIVE_POOL_FLOOR; tuned by timing a real
* large-repo `analyze --force` at this factor vs a forced 2 GiB pool (the pool
* is a native eager allocation, measured with a real analyze, not a build-free
* bench — see the emit-path COPY timing note in bench/emit-persistence).
*/
const POOL_BYTES_PER_ELEMENT = 4 * 1024;
/**
* Size the buffer pool to an estimated graph size (node + relationship count),
* clamped to [ADAPTIVE_POOL_FLOOR, scaledAnalyzePoolCap], with every term
* scaled by granuleRatio (#2631): on non-4K hosts the engine bills pool
* budget per OS-page-sized granule, so the same graph consumes up to
* pageSize/4096 × the budget it needs on x86. On 4 KiB hosts the ratio is 1
* and this is byte-identical to the pre-#2631 behavior. The estimate is never
* above the page-size-scaled #2557 cap bounded by 80% of RAM, never below the
* scaled COPY-safety floor; the hintless default stays unscaled.
*
* `pageSize` is a test seam (the pageSizeDoctorLines convention); production
* callers omit it and get the memoized real OS page size.
*/
export const estimateBufferPool = (
graphElementCount: number,
pageSize: number | undefined = getOsPageSize(),
): number =>
clampBufferPool(graphElementCount * POOL_BYTES_PER_ELEMENT * granuleRatio(pageSize), pageSize);
/**
* Optional per-run buffer-pool size hint (bytes). The analyze orchestrator sets
* it once the graph size is known (after the pipeline, before the DB open) so
* the pool is sized to the repo instead of the fixed 2 GiB default, and clears
* it at run end. Non-analyze opens (MCP serve, `native-check` `:memory:`) never
* set it and keep the default.
*/
let bufferPoolSizeHint: number | undefined;
/** Set (bytes) or clear (`undefined`) the per-run buffer-pool size hint. */
export const setBufferPoolSizeHint = (bytes: number | undefined): void => {
bufferPoolSizeHint = bytes;
};
/**
* Resolve the `bufferManagerSize` passed to every `new lbug.Database(...)`.
* `GITNEXUS_LBUG_BUFFER_POOL_SIZE` (bytes) overrides everything; `0` is a
* `GITNEXUS_LBUG_BUFFER_POOL_SIZE` (bytes) overrides the default; `0` is a
* deliberate escape hatch that restores LadybugDB's native unbounded
* 80%-of-RAM default. With no env override, a per-run `setBufferPoolSizeHint`
* (clamped to [floor, default]) sizes the pool to the repo; otherwise the
* default. Resolved at call time (not module load) so tests can stub the env
* var, the hint, and `os.totalmem`.
* 80%-of-RAM default. Resolved at call time (not module load) so tests can
* stub the env var and `os.totalmem`.
*/
const resolveBufferManagerSize = (): number => {
const raw = process.env.GITNEXUS_LBUG_BUFFER_POOL_SIZE;
if (raw === undefined) {
return bufferPoolSizeHint !== undefined
? clampBufferPool(bufferPoolSizeHint)
: defaultBufferPoolSize();
}
if (raw === undefined) return defaultBufferPoolSize();
const parsed = parseBufferPoolSize(raw);
if (parsed !== undefined) return parsed;
// Non-empty but unparseable input: warn the operator and fall back —
@@ -492,64 +350,12 @@ const resolveBufferManagerSize = (): number => {
if (raw.trim().length > 0) {
logger.warn(
{ rawValue: raw, fallback: defaultBufferPoolSize() },
`Ignoring invalid GITNEXUS_LBUG_BUFFER_POOL_SIZE=${raw}; expected integer >= 0 (bytes; 0 restores the native 80%-of-RAM default); falling back to the platform default pool size.`,
`Ignoring invalid GITNEXUS_LBUG_BUFFER_POOL_SIZE=${raw}; expected integer >= 0 (bytes; 0 restores the native 80%-of-RAM default); falling back to min(2 GiB, 80% of RAM).`,
);
}
return defaultBufferPoolSize();
};
/**
* Doctor-facing view of the pool size the next Database open would get
* (#2631): env override > clamped hint > unscaled hintless default. Read-only;
* doctor prints it next to the page-size lines so support triage sees the
* sizing inputs at a glance. `0` is the pass-through sentinel for LadybugDB's
* native 80%-of-RAM default — callers must label it, not print "0 MiB".
*/
export const getEffectiveBufferPoolSize = (): number => resolveBufferManagerSize();
/**
* Matches the engine's buffer-pool exhaustion throw (buffer_manager.cpp:
* "Unable to allocate memory! The buffer pool is full and no memory could be
* freed!"). Distinct from isLbugPageSizeFrameError above, which matches the
* madvise/frame-release failure class.
*/
const BUFFER_POOL_EXHAUSTION_RE = /buffer pool is full|unable to allocate memory/i;
const formatMiB = (bytes: number): string => `${Math.round(bytes / (1024 * 1024))} MiB`;
/**
* Actionable remedy for a buffer-pool exhaustion error (#2631), or undefined
* when `message` is not that class. Cause → consequence → remedy, the
* diagnoseExtensionLoad convention: names the effective pool, the override
* knob, and — on non-4K hosts — the granule amplification that makes the
* budget exhaust early (the reporter's Ascend/aarch64 64 KiB kernel billed a
* pool up to 16× faster than the same analyze on x86).
*/
export const bufferPoolExhaustionRemedy = (
message: string,
pageSize: number | undefined = getOsPageSize(),
): string | undefined => {
if (!BUFFER_POOL_EXHAUSTION_RE.test(message)) return undefined;
const ratio = granuleRatio(pageSize);
const pool = resolveBufferManagerSize();
// 0 is the pass-through sentinel (GITNEXUS_LBUG_BUFFER_POOL_SIZE=0 →
// LadybugDB's native 80%-of-RAM default) — "0 MiB" would be nonsense in the
// very triage text this remedy exists to provide.
const poolLabel = pool === 0 ? "LadybugDB's native 80%-of-RAM default" : formatMiB(pool);
const pageNote =
ratio > 1
? ` This host's ${(pageSize ?? 0) / 1024} KiB OS page size makes the engine bill pool ` +
`memory in ${(pageSize ?? 0) / 1024} KiB granules — up to ${ratio}× faster budget use ` +
`than a 4 KiB-page host running the same analyze.`
: '';
return (
`The LadybugDB buffer pool (${poolLabel}) was exhausted during the bulk COPY.` +
pageNote +
` Set GITNEXUS_LBUG_BUFFER_POOL_SIZE=<bytes> to raise it (e.g. ${4 * 1024 * 1024 * 1024}` +
` for 4 GiB); 0 restores LadybugDB's native 80%-of-RAM default.`
);
};
/** Matches WAL corruption errors from the LadybugDB engine. */
const WAL_CORRUPTION_RE = /corrupt(ed)?\s+wal|invalid\s+wal\s+record|wal.*corrupt|checksum.*wal/i;
@@ -636,12 +442,8 @@ const LBUG_PAGE_COMBO_RE = /unsupported page size combination/i;
* True when `err` looks like the LadybugDB buffer manager failing to release
* frame memory — the failure mode of a 4 KiB page-size assumption on a
* 16 KiB/64 KiB-page kernel (#1231). Deliberately does NOT match the
* generic "buffer pool is full" exhaustion error: that one is handled as a
* SIZING problem — though since #2631 we know page size drives sizing too
* (the engine bills pool budget per OS-page-sized discard granule, so non-4K
* hosts exhaust the same budget up to pageSize/4096× earlier; see
* granuleRatio, which scales the pool accordingly, and
* bufferPoolExhaustionRemedy, which explains it to the operator).
* generic "buffer pool is full" exhaustion error, which is a sizing
* problem, not a page-size one.
*/
export const isLbugPageSizeFrameError = (err: unknown): boolean => {
if (!err) return false;
@@ -668,16 +470,6 @@ export const isPageSizeAwareLadybug = (version: string | undefined): boolean =>
// because analyze error paths and doctor may both ask, and getconf forks.
let cachedOsPageSize: number | null | undefined;
/**
* Test seam (the `_captureLogger` convention): pin the memoized OS page size
* so sizing tests are host-independent — without this they would silently
* drift on 16 KiB-page Apple Silicon runners. `number` pins a value, `null`
* pins "undetectable", `undefined` clears the memo so the next call re-probes.
*/
export const _setOsPageSizeForTests = (pageSize: number | null | undefined): void => {
cachedOsPageSize = pageSize;
};
/**
* OS memory page size in bytes, or `undefined` when it cannot be determined
* (Windows, missing getconf, sandboxed exec). Node exposes no page-size API,
@@ -716,6 +508,11 @@ export const getOsPageSize = (): number | undefined => {
return cachedOsPageSize ?? undefined;
};
/** Exported only for unit tests — clears the getconf probe cache. */
export const _resetOsPageSizeCacheForTest = (): void => {
cachedOsPageSize = undefined;
};
type LbugModule = typeof lbug;
export interface LbugDatabaseOptions {
@@ -764,29 +561,6 @@ export const isDbBusyError = (err: unknown): boolean => {
);
};
/**
* True when a WAL-checkpoint IO error ALSO carries a busy/lock signal — the
* rotation failed because another handle (a `gitnexus mcp` server, or this
* process's own reader) holds the store's WAL open, rather than a permanent
* disk error. Reuses `isDbBusyError`'s already-tested keyword set instead of a
* fresh regex, so an unmatched message degrades to "IO error" rather than
* silently claiming a held-open cause. (#2599)
*/
export const isLbugCheckpointBusyError = (err: unknown): boolean => {
if (!isLbugCheckpointIoError(err)) return false;
// Anchor to real held-open wording rather than isDbBusyError's broad
// `.includes('lock')`, which matches the DB PATH embedded in the checkpoint
// error message (e.g. a repo under `blockchain-app`) and would misclassify a
// pure disk fault as held-open (#2614 LOW).
const msg = (err instanceof Error ? err.message : String(err)).toLowerCase();
return (
msg.includes('could not set lock') ||
msg.includes('lock is held') ||
msg.includes('being used by another process') ||
msg.includes('is busy')
);
};
/** See {@link classifyDeleteAllError}. */
export type DeleteAllErrorClass = 'benign-missing-table' | 'rethrow';
@@ -874,19 +648,6 @@ export const HANDLE_RELEASE_PROBE_ATTEMPTS = 5;
export const HANDLE_RELEASE_PROBE_DELAY_MS = 50;
const HANDLE_RELEASE_LOCK_CODES = new Set(['EBUSY', 'EPERM', 'EACCES']);
// Retry-budget registry, part 2 (retry-budget consolidation): the remaining
// open-time lock retries live next to their call sites but are catalogued here
// so all lbug retry budgets surface in one grep. They retry the same lock class
// as 1–3 ("Could not set lock" while a writer rebuilds the index):
// 4. LOCK_RETRY_ATTEMPTS / LOCK_RETRY_DELAY_MS (pool-adapter.ts)
// → read pool's read-only open while `gitnexus analyze` is writing
// (3 attempts, linear 2s·n back-off ≈ 6s total)
// 5. LBUG_OPEN_RETRY_ATTEMPTS / _BASE_MS / _MAX_MS (group/bridge-db.ts)
// → cross-repo bridge RO open race (10 attempts, linear 100ms·n capped
// at 500ms ≈ 3.5s total)
// Kept in-file (not moved here) so explicit `lbug-config` test mocks don't have
// to enumerate them; change a budget in its call site and update this catalogue.
/**
* Test-fixture directory prefixes recognized by `isTestFixturePath`.
*
+3 -37
View File
@@ -96,9 +96,6 @@ export interface FtsProbeResult {
reason?: string;
}
/** Same shape for every optional extension; `FtsProbeResult` is the legacy name. */
export type ExtensionProbeResult = FtsProbeResult;
const DEFAULT_FTS_PROBE_TIMEOUT_MS = 10_000;
/** A LadybugDB query result exposes a synchronous `close()`. */
@@ -139,39 +136,8 @@ const closeProbeResults = (result: unknown): void => {
export async function probeFtsExtensionLoad(
timeoutMs: number = DEFAULT_FTS_PROBE_TIMEOUT_MS,
): Promise<FtsProbeResult> {
return await probeExtensionLoad('fts', timeoutMs);
}
/**
* Live-probe `LOAD EXTENSION vector`, the VECTOR counterpart of the FTS probe.
*
* Needed for the same reason #2374 needed the FTS one, and reported the same
* way: #2623's reporter saw `doctor` print `VECTOR index: available` while
* every incremental `analyze` was dying because the extension had not loaded.
* `doctor` derived that line from a static platform capability, so it read
* "available" no matter what the extension file was doing.
*
* Probes for real on every platform, Windows included: the extension server
* ships win_amd64 VECTOR artifacts for every 0.18.x extension version (the
* old blanket Windows refusal was stale, #1365-era). LOAD never touches the
* network and never invokes the installer, so this probe is exactly as safe
* as the FTS one above.
*/
export async function probeVectorExtensionLoad(
timeoutMs: number = DEFAULT_FTS_PROBE_TIMEOUT_MS,
): Promise<ExtensionProbeResult> {
return await probeExtensionLoad('vector', timeoutMs);
}
/**
* Shared LOAD probe. `extension` is a fixed internal literal, never user input.
*/
async function probeExtensionLoad(
extension: 'fts' | 'vector',
timeoutMs: number,
): Promise<ExtensionProbeResult> {
let timer: ReturnType<typeof setTimeout> | undefined;
const timeout = new Promise<ExtensionProbeResult>((resolve) => {
const timeout = new Promise<FtsProbeResult>((resolve) => {
timer = setTimeout(
() =>
resolve({
@@ -182,7 +148,7 @@ async function probeExtensionLoad(
);
});
const probe = (async (): Promise<ExtensionProbeResult> => {
const probe = (async (): Promise<FtsProbeResult> => {
try {
const { default: lbug } = await import('@ladybugdb/core');
const db = new lbug.Database(':memory:');
@@ -190,7 +156,7 @@ async function probeExtensionLoad(
try {
const conn = new lbug.Connection(db);
try {
const result = await conn.query(`LOAD EXTENSION ${extension}`);
const result = await conn.query('LOAD EXTENSION fts');
closeProbeResults(result);
return { loaded: true };
} finally {
+7 -133
View File
@@ -17,7 +17,7 @@
import fs from 'fs/promises';
import lbug from '@ladybugdb/core';
import { isReadOnlyDbError, loadFTSExtension, loadVectorExtension } from './lbug-adapter.js';
import { isReadOnlyDbError, loadFTSExtension } from './lbug-adapter.js';
import { closeQueryResults } from './query-result-utils.js';
import {
createLbugDatabase,
@@ -53,44 +53,10 @@ interface PoolEntry {
}>;
lastUsed: number;
dbPath: string;
/** Filesystem identity of the on-disk DB at open time. When `analyze`
* rebuilds or mutates the index, this diverges from the current file and
* initLbug re-opens the pool onto the new file instead of serving the
* stale open inode. Null for injected/external databases (initLbugWithDb),
* which are never invalidated this way. */
dbIdentity: DbIdentity | null;
/** Set to true when the pool entry is closed — checkin will close orphaned connections */
closed: boolean;
}
/** Filesystem identity used to detect an index rebuilt/mutated under a live
* read pool. `ino` catches a full-rebuild unlink+recreate or an atomic-rename
* swap; `mtimeMs`+`size` catch an in-place incremental writeback. */
interface DbIdentity {
ino: number;
mtimeMs: number;
size: number;
}
export async function statDbIdentity(dbPath: string): Promise<DbIdentity | null> {
try {
const s = await fs.stat(dbPath);
return { ino: s.ino, mtimeMs: s.mtimeMs, size: s.size };
} catch {
return null;
}
}
/** True only when both identities are known AND differ. A stat failure
* (ENOENT during the brief unlink window of a full rebuild) yields false, so
* the reader keeps serving its still-valid open inode until the NEW file
* appears with a different identity — avoiding a churn into a failed reopen
* mid-rebuild. */
export function dbIdentityChanged(prev: DbIdentity | null, next: DbIdentity | null): boolean {
if (!prev || !next) return false;
return prev.ino !== next.ino || prev.mtimeMs !== next.mtimeMs || prev.size !== next.size;
}
const pool = new Map<string, PoolEntry>();
/**
@@ -126,18 +92,6 @@ interface SharedDB {
db: lbug.Database;
refCount: number;
ftsLoaded: boolean;
/** VECTOR loaded on this Database. Extension load scope is per-Database
* (probe-verified on @ladybugdb/core 0.18.x): loading on any one
* connection enables QUERY_VECTOR_INDEX on every connection of the same
* Database. Without this load the pool's vector lane raised a Catalog
* exception on every semantic query and silently fell back to the exact
* scan (#2623 follow-up). Optional with `?? false` semantics so the
* construction sites stay minimal. */
vectorLoaded?: boolean;
/** File identity at open — used to detect reuse of a shared read-only handle
* whose on-disk index was rebuilt/swapped since it opened (only reachable
* when a second pool consumer shares this dbPath; #2614 F2). */
dbIdentity?: DbIdentity | null;
/** When true, closeOne skips db.close() — the Database is owned externally. */
external?: boolean;
}
@@ -366,7 +320,6 @@ function closeOne(repoId: string): void {
// for the same dbPath reuse it instead of hitting a file lock.
shared.refCount = 0;
shared.ftsLoaded = false;
shared.vectorLoaded = false;
} else {
shared.db.close().catch(() => {});
dbCache.delete(entry.dbPath);
@@ -436,16 +389,7 @@ setInterval(() => {
function createConnection(db: lbug.Database): lbug.Connection {
silenceStdout();
try {
const conn = new lbug.Connection(db);
// Bound a single query at the engine level so a pathological query cannot
// hang a pooled connection past the JS-side Promise.race guard (which frees
// the waiter but not the native call). Matches QUERY_TIMEOUT_MS. Guarded so
// test doubles that don't model the engine method don't break connection
// creation.
if (typeof conn.setQueryTimeout === 'function') {
conn.setQueryTimeout(QUERY_TIMEOUT_MS);
}
return conn;
return new lbug.Connection(db);
} finally {
restoreStdout();
}
@@ -456,8 +400,6 @@ const QUERY_TIMEOUT_MS = 30_000;
/** Waiter queue timeout in milliseconds */
const WAITER_TIMEOUT_MS = 15_000;
// Read-only open retry while `gitnexus analyze` writes. Catalogued as entry 4
// of the lbug-config retry-budget registry.
const LOCK_RETRY_ATTEMPTS = 3;
const LOCK_RETRY_DELAY_MS = 2000;
const SHADOW_REPLAY_PROBE_QUERY = 'MATCH (n) RETURN n LIMIT 1';
@@ -651,45 +593,18 @@ const initPromises = new Map<string, Promise<void>>();
* Concurrent calls for the same repoId are deduplicated — the second caller
* awaits the first's in-progress init rather than starting a redundant one.
*/
/**
* Returns `true` when this call (re)opened a fresh handle onto the current
* on-disk file, `false` when it reused/served the existing handle (unchanged,
* or changed-but-a-query-is-in-flight). Callers that gate their own freshness
* bookkeeping on "did the pool actually roll over" (LocalBackend) use the
* return value; callers that only need the pool ready can ignore it.
*/
export const initLbug = async (repoId: string, dbPath: string): Promise<boolean> => {
export const initLbug = async (repoId: string, dbPath: string): Promise<void> => {
const existing = pool.get(repoId);
if (existing) {
existing.lastUsed = Date.now();
// Detect an index that `analyze` rebuilt or mutated under this live read
// pool. Without this, the pool keeps serving the old (POSIX:
// unlinked-but-open) inode until LRU/idle eviction — a stale-read window
// of up to IDLE_TIMEOUT_MS after analyze finishes.
const current = await statDbIdentity(dbPath);
if (!dbIdentityChanged(existing.dbIdentity, current)) return false; // unchanged → reuse
// A query is in flight on this entry; closing its connection (and the
// shared Database at refCount 0) mid-use is a native use-after-free. Serve
// the current handle for this dispatch — the next initLbug that finds the
// entry idle (checkedOut === 0) reopens, since the identity stays divergent
// until then. Under sustained overlapping queries `checkedOut` may never
// reach 0 and `lastUsed` keeps the idle timer from evicting, so this window
// is bounded by the load, not IDLE_TIMEOUT_MS — the data stays consistent
// (a complete older snapshot), just not the newest. Callers that route
// freshness THROUGH initLbug (rather than calling closeLbug directly) get
// this guard for free; that is why LocalBackend delegates here (#2614).
if (existing.checkedOut > 0) return false;
closeOne(repoId); // idle & changed → evict, then fall through to reopen the new file
return;
}
// Deduplicate concurrent init calls for the same repoId —
// prevents double-init race when multiple parallel tool calls
// trigger initialization for the same repo simultaneously.
const pending = initPromises.get(repoId);
if (pending) {
await pending;
return true;
}
if (pending) return pending;
const promise = doInitLbug(repoId, dbPath);
initPromises.set(repoId, promise);
@@ -698,7 +613,6 @@ export const initLbug = async (repoId: string, dbPath: string): Promise<boolean>
} finally {
initPromises.delete(repoId);
}
return true;
};
/**
@@ -719,23 +633,6 @@ async function doInitLbug(repoId: string, dbPath: string): Promise<void> {
// Reuse an existing native Database if another repoId already opened this path.
// This prevents buffer manager exhaustion from multiple mmap regions on the same file.
let shared = dbCache.get(dbPath);
if (shared && !shared.external && shared.dbIdentity) {
// #2614 F2: a cached read-only Database is keyed by dbPath and shared across
// pool consumers. If the on-disk index was rebuilt/swapped (new inode) while
// ANOTHER consumer still holds this handle (refCount kept it alive), reusing
// it serves a superseded index. Unreachable via the MCP backend (one
// consumer per lbugPath ⇒ refCount hits 0 ⇒ closeOne reopens fresh); a
// complete fix needs per-inode handles rather than a dbPath-keyed cache.
// Surface it so the corner is observable instead of silently stale.
const current = await statDbIdentity(dbPath);
if (dbIdentityChanged(shared.dbIdentity, current)) {
realStderrWrite(
`GitNexus: reusing a shared read-only handle for ${dbPath} whose on-disk ` +
`index was rebuilt while another consumer holds it — results may be stale ` +
`until that consumer releases it.\n`,
);
}
}
if (!shared) {
// Open in read-only mode — MCP server never writes to the database.
// This allows multiple MCP server instances to read concurrently, and
@@ -744,7 +641,7 @@ async function doInitLbug(repoId: string, dbPath: string): Promise<void> {
for (let attempt = 1; attempt <= LOCK_RETRY_ATTEMPTS; attempt++) {
try {
const db = await openReadOnlyDatabase(dbPath);
shared = { db, refCount: 0, ftsLoaded: false, dbIdentity: await statDbIdentity(dbPath) };
shared = { db, refCount: 0, ftsLoaded: false };
dbCache.set(dbPath, shared);
break;
} catch (err: any) {
@@ -753,12 +650,7 @@ async function doInitLbug(repoId: string, dbPath: string): Promise<void> {
if (isWalCorruptionError(lastError)) {
try {
const db = await tryQuarantineAndReopen(dbPath, repoId);
shared = {
db,
refCount: 0,
ftsLoaded: false,
dbIdentity: await statDbIdentity(dbPath),
};
shared = { db, refCount: 0, ftsLoaded: false };
dbCache.set(dbPath, shared);
break;
} catch (retryErr) {
@@ -819,20 +711,10 @@ async function doInitLbug(repoId: string, dbPath: string): Promise<void> {
if (!shared.ftsLoaded) {
shared.ftsLoaded = await loadFTSExtension(available[0], { policy: 'load-only' });
}
// VECTOR too — extension load scope is per-Database, so this one load
// makes QUERY_VECTOR_INDEX legal on every pooled connection. Same
// load-only contract as FTS above; on failure the semantic-query lane
// falls back to the exact scan with its own diagnostic (#2623 follow-up).
if (!shared.vectorLoaded) {
shared.vectorLoaded = await loadVectorExtension(available[0], { policy: 'load-only' });
}
// Register pool entry only after all connections are pre-warmed and FTS is
// loaded. Concurrent executeQuery calls see either "not initialized"
// (and throw cleanly) or a fully ready pool — never a half-built one.
// Record the on-disk identity so a later initLbug can detect an analyze
// rebuild/mutation and re-open onto the new file (pool staleness invalidation).
const dbIdentity = await statDbIdentity(dbPath);
pool.set(repoId, {
db,
available,
@@ -840,7 +722,6 @@ async function doInitLbug(repoId: string, dbPath: string): Promise<void> {
waiters: [],
lastUsed: Date.now(),
dbPath,
dbIdentity,
closed: false,
});
ensureIdleTimer();
@@ -896,11 +777,6 @@ export async function initLbugWithDb(
if (!shared.ftsLoaded) {
shared.ftsLoaded = await loadFTSExtension(available[0], { policy: 'load-only' });
}
// VECTOR too — same per-Database scope and load-only contract as the
// doInitLbug site above (#2623 follow-up).
if (!shared.vectorLoaded) {
shared.vectorLoaded = await loadVectorExtension(available[0], { policy: 'load-only' });
}
pool.set(repoId, {
db: existingDb,
@@ -909,8 +785,6 @@ export async function initLbugWithDb(
waiters: [],
lastUsed: Date.now(),
dbPath,
// Injected/external DB (tests) — not tracked for rebuild invalidation.
dbIdentity: null,
closed: false,
});
ensureIdleTimer();
-1
View File
@@ -402,7 +402,6 @@ CREATE REL TABLE ${REL_TABLE_NAME} (
FROM \`Static\` TO Community,
FROM \`Variable\` TO Community,
FROM \`Property\` TO Community,
FROM \`Property\` TO \`Property\`,
FROM \`Record\` TO Method,
FROM \`Record\` TO \`Constructor\`,
FROM \`Record\` TO \`Property\`,
+1 -1
View File
@@ -60,7 +60,7 @@ export const isMissingFsError = (err: unknown): boolean =>
const missing = isMissingFsError;
export const sidecarPreflightDisabled = (): boolean =>
const sidecarPreflightDisabled = (): boolean =>
/^(1|true|yes|on)$/i.test(process.env.GITNEXUS_DISABLE_LBUG_SIDECAR_PREFLIGHT ?? '');
export const statIfExists = async (filePath: string): Promise<{ size: number } | null> => {
@@ -118,9 +118,6 @@ export const runCheckpointWithRetry = async (
{ attempts: CHECKPOINT_RETRY_ATTEMPTS },
'GitNexus: manual WAL checkpoint exhausted retry budget — surfacing IO error to caller',
);
// The held-open cause (#2599) is named at the CLI layer (analyze.ts) where the
// --wal-checkpoint-threshold recovery hint already renders, so the original IO
// error is preserved intact for that classifier rather than re-wrapped here.
throw lastError;
};
+11 -12
View File
@@ -86,24 +86,23 @@ export const getRuntimeFingerprint = (): RuntimeFingerprint => ({
onnxruntime: packageVersion('onnxruntime-node'),
});
export const isVectorExtensionSupportedByPlatform = (
platform: NodeJS.Platform = process.platform,
): boolean => platform !== 'win32';
export const getRuntimeCapabilities = (): RuntimeCapabilities => {
const vector = isVectorExtensionSupportedByPlatform() ? 'available' : 'unavailable';
const exactScanLimit = getExactScanLimit();
// Static PLATFORM capability only. LadybugDB ships the VECTOR extension for
// every platform gitnexus supports — the extension server hosts win_amd64
// artifacts for every 0.18.x extension version (probed: v0.18.0 and v0.18.1
// both return a real 14 MB PE32+ DLL; the pinned 0.18.2 core resolves its
// extension directory to 0.18.1, strace-verified), so the old
// `platform !== 'win32'` gate was stale (#1365-era). Whether the extension
// actually LOADS on a given machine is a runtime question — doctor answers
// it with probeVectorExtensionLoad, and analyze/query degrade to exact scan
// when the load fails.
return {
graph: 'available',
fts: 'available',
vector: 'available',
semanticMode: 'vector-index',
vector,
semanticMode: vector === 'available' ? 'vector-index' : 'exact-scan',
exactScanLimit,
reason: undefined,
reason:
vector === 'unavailable'
? 'LadybugDB VECTOR is disabled on this platform; semantic search uses exact scan when embeddings exist.'
: undefined,
};
};
+17 -217
View File
@@ -11,7 +11,6 @@
import path from 'path';
import fs from 'fs/promises';
import { retryRename } from '../storage/fs-atomic.js';
import { runPipelineFromRepo } from './ingestion/pipeline.js';
import type { KnowledgeGraph } from './graph/types.js';
import { resetDegradedParseCounter } from './tree-sitter/safe-parse.js';
@@ -25,7 +24,6 @@ import {
closeLbugBeforeExit,
loadCachedEmbeddings,
deleteNodesForFiles,
ensureEmbeddingRowDmlSafe,
deleteAllCommunitiesAndProcesses,
deleteAllInterprocTaintPaths,
deleteAllCallSummaries,
@@ -36,12 +34,10 @@ import {
LbugWipeError,
DELETE_FILES_CHUNK_SIZE,
} from './lbug/lbug-adapter.js';
import { estimateBufferPool, setBufferPoolSizeHint } from './lbug/lbug-config.js';
import { escapeCypherString } from './lbug/cypher-escape.js';
import {
buildSearchIndexesOrDegrade,
createSearchFTSIndexes,
dropSearchFTSIndexes,
initialiseSearchFTSStemmer,
verifySearchFTSIndexes,
} from './search/fts-indexes.js';
@@ -57,10 +53,7 @@ import {
checkpointOnce,
type WalCheckpointDriver,
} from './lbug/wal-checkpoint-driver.js';
import {
quarantineSidecarsForDirtyRecovery,
inspectLbugSidecars,
} from './lbug/sidecar-recovery.js';
import { quarantineSidecarsForDirtyRecovery } from './lbug/sidecar-recovery.js';
import type { EmbeddingIdentity } from './embeddings/embedding-identity.js';
import {
getStoragePaths,
@@ -130,7 +123,6 @@ import { EMBEDDING_TABLE_NAME } from './lbug/schema.js';
import { STALE_HASH_SENTINEL } from './lbug/schema.js';
import { isSpringBeanCandidateSourceFile } from './ingestion/frameworks/spring/bean-catalog.js';
import { SPRING_BEAN_INVENTORY_FEATURE } from './ingestion/frameworks/spring/analysis-features.js';
import { SPRING_CONFIG_BINDINGS_FEATURE } from './ingestion/languages/java/analysis-features.js';
import {
CLASS_FRAMEWORK_ANNOTATIONS_FEATURE,
findAnalysisFeatureMismatches,
@@ -145,7 +137,6 @@ import {
const ANALYSIS_FEATURES = [
CLASS_FRAMEWORK_ANNOTATIONS_FEATURE,
SPRING_BEAN_INVENTORY_FEATURE,
SPRING_CONFIG_BINDINGS_FEATURE,
] as const;
interface PersistedFrameworkAnnotationRow {
@@ -654,11 +645,6 @@ export async function runFullAnalysis(
// and are shared across branches (#2106 KTD7).
const { storagePath } = getStoragePaths(repoPath);
// Start each analyze with a clean buffer-pool hint: any pre-pipeline DB open
// (e.g. the embeddings-cache open) falls back to the default until the hint is
// set from the built graph below, so a prior run's size can't leak in.
setBufferPoolSizeHint(undefined);
// Clean up stale KuzuDB files from before the LadybugDB migration.
const kuzuResult = await cleanupOldKuzuFiles(storagePath);
if (kuzuResult.found && kuzuResult.needsReindex) {
@@ -1334,54 +1320,6 @@ export async function runFullAnalysis(
? diffFileHashes(newFileHashes, existingMeta!.fileHashes)
: undefined;
// #2 atomic index publish: on a full rebuild, build the fresh DB at a temp
// path and swap it over the live index in one rename at the very end, so a
// concurrent MCP reader opening mid-build only ever sees the previous
// complete index (never a wiped/half-built file) and a crash leaves the old
// index intact. The whole build flows through the singleton connection, so
// only initLbug/wipeLbugDbFiles below take the temp target.
//
// POSIX only: the common CLI/serve-worker analyze paths skip the native close
// (closeLbugBeforeExit, #2264) and leave the build handle open at swap time.
// POSIX renames an open file cleanly; a same-process open handle blocks the
// rename on Windows. Windows keeps the current in-place behavior
// (buildPath === lbugPath, no swap) until that is resolved (see §12/follow-up).
const isFullRebuild = !(isIncremental && hashDiff);
// Where the swap is allowed:
// - POSIX renames an open file, so the usual skip-native-close (#2264) is
// fine and the swap always applies.
// - Windows can swap only when a real close is safe to release the build
// handle before the rename — i.e. NOT a --pdg run (the #2264 destructor
// crash). Unverified on Windows CI; falls back to in-place otherwise.
const posixSwap = process.platform !== 'win32';
// #2614 Windows: the forced real-close before the rename re-bets that #2264 is
// --pdg-only, which is unproven (the CLI/worker skip the native close
// UNCONDITIONALLY) and unverifiable without a Windows runner. Keep it opt-in
// (GITNEXUS_ATOMIC_WINDOWS_SWAP=1) so the default Windows analyze stays on the
// proven in-place path; enable it only to test the Windows swap.
const windowsSwapOk =
process.platform === 'win32' &&
options.pdg !== true &&
process.env.GITNEXUS_ATOMIC_WINDOWS_SWAP === '1';
// Incremental atomicity copies the whole index into the temp before mutating
// it, which negates incremental's speed premise — so it is opt-in
// (GITNEXUS_ATOMIC_INCREMENTAL=1) pending a benchmark. Full rebuilds always
// swap where the platform allows.
const wantAtomicIncremental =
isIncremental && !!hashDiff && process.env.GITNEXUS_ATOMIC_INCREMENTAL === '1';
// #2614 F3: the copy-then-swap stages ONLY the main lbug file, so a live index
// carrying an orphan .wal/.shadow (a silently-failed prior checkpoint) would
// be copied incompletely and lose that delta. Only take the atomic path when
// the live index is a consolidated single file; otherwise fall back to the
// in-place writeback, which the next open replays correctly.
const atomicIncremental =
wantAtomicIncremental && (await inspectLbugSidecars(lbugPath)).kind === 'clean';
if (wantAtomicIncremental && !atomicIncremental) {
log('atomic-incremental: live index carries orphan sidecars — using in-place writeback');
}
const useAtomicSwap = (isFullRebuild || atomicIncremental) && (posixSwap || windowsSwapOk);
const buildPath = useAtomicSwap ? `${lbugPath}.new` : lbugPath;
if (isIncremental && hashDiff) {
log(
`Incremental: changed=${hashDiff.changed.length}, ` +
@@ -1404,14 +1342,6 @@ export async function runFullAnalysis(
directWriteCount: hashDiff.toWrite.length,
},
});
if (atomicIncremental) {
// Stage the live index into the temp so the in-place delete/writeback
// below mutates the COPY, and the end-of-run swap publishes it atomically.
// Clear any stale temp first (a crashed run), then copy the (consolidated,
// single-file) live index. Whole-file copy — hence opt-in.
await wipeLbugDbFiles(buildPath);
await fs.copyFile(lbugPath, buildPath);
}
} else {
// Full rebuild path: wipe DB files first.
// Set the dirty flag BEFORE the wipe whenever a prior meta exists,
@@ -1441,27 +1371,10 @@ export async function runFullAnalysis(
// valve below can never drift. Failures now throw a typed LbugWipeError
// (ENOENT-verified removal) instead of silently letting initLbug reopen
// a still-populated DB this run believes it wiped.
//
// With the atomic swap (POSIX), this wipes the TEMP build target
// (`buildPath` = `<lbugPath>.new`, clearing any stragglers from a crashed
// run) and leaves the live index untouched until the end-of-run swap. On
// Windows buildPath === lbugPath, so this is the original in-place wipe.
await wipeLbugDbFiles(buildPath);
await wipeLbugDbFiles(lbugPath);
}
// Size the buffer pool to the graph just built by the pipeline (a page cache
// over the on-disk index, which scales with node/edge count) instead of the
// fixed 2 GiB default, whose eager commit dominates large-repo analyze. The
// size is clamped to [COPY-safety floor, default], so it only ever shrinks
// the pool; env override / no-hint paths are unchanged. See
// resolveBufferManagerSize / estimateBufferPool.
setBufferPoolSizeHint(
estimateBufferPool(pipelineResult.graph.nodeCount + pipelineResult.graph.relationshipCount),
);
// Full rebuild (POSIX) builds into the temp `buildPath`; incremental and
// Windows use `buildPath === lbugPath` in place.
await initLbug(buildPath);
await initLbug(lbugPath);
// Manual WAL checkpoint driver (#1741): periodically drain the WAL
// from JS so the un-retriable native auto-checkpoint almost never
@@ -1687,44 +1600,7 @@ export async function runFullAnalysis(
// DB write plan changes here; fileHashes/meta bookkeeping is identical.
// Thresholds + the AND-gate live in incremental/escalation-gate.ts.
const writeFraction = effectiveWriteSet.size / Math.max(1, allFilePaths.length);
// VECTOR gate (#2623) — load the extension BEFORE a single embedding row
// is touched. `deleteNodesForFiles` below opens with the CodeEmbedding
// join-delete, and LadybugDB refuses all DML on a table carrying its HNSW
// index unless VECTOR is loaded on this connection; nothing else on this
// path loads it until Phase 4, so every incremental run over a DB that
// already built `code_embedding_idx` died here. Same seam the FTS drop
// occupies at the head of this branch (#2589): index lifecycle first,
// then rows. UNCONDITIONAL — not gated on `shouldGenerateEmbeddings` —
// because a DB carrying the index from an earlier `--embeddings` run hits
// the identical wall on a plain incremental run.
//
// When VECTOR genuinely cannot load, the table is immutable (the index
// cannot be dropped without the extension either), so surgery is
// impossible: fall through to the escalation valve's wipe-and-COPY plan,
// which rebuilds the DB files outright and needs no embedding-row DML.
const embeddingRowDmlSafe = await ensureEmbeddingRowDmlSafe();
if (!embeddingRowDmlSafe && cachedEmbeddings.length === 0) {
// The escalation below WIPES the DB files, and Phase 3.5 restores
// embedding rows from `cachedEmbeddings` — which is only populated when
// `deriveEmbeddingMode` saw `meta.stats.embeddings > 0`. A DB whose meta
// under-reports its embeddings (meta restored from an older run, or a
// count that never got stamped) would therefore have every vector
// silently destroyed by a rebuild it did not ask for. Read them now,
// while the DB is still intact — a plain MATCH, which needs no VECTOR
// extension. Rows whose owning node is gone are dropped by Phase 3.5's
// live-graph filter, exactly as on any other wiped path.
const rescued = await loadCachedEmbeddings();
if (rescued.embeddings.length > 0) {
cachedEmbeddings = rescued.embeddings;
cachedEmbeddingNodeIds = rescued.embeddingNodeIds;
log(
`Preserving ${rescued.embeddings.length} embedding row(s) across the forced rebuild ` +
`(the index metadata did not account for them).`,
);
}
}
if (
!embeddingRowDmlSafe ||
shouldEscalateIncrementalWrite(
filesToDelete.length,
effectiveWriteSet.size,
@@ -1733,20 +1609,13 @@ export async function runFullAnalysis(
) {
escalatedFullWrite = true;
log(
!embeddingRowDmlSafe
? `Incremental: the ${EMBEDDING_TABLE_NAME} vector index exists but the VECTOR ` +
`extension could not be loaded, so embedding rows cannot be rewritten in place — ` +
`switching to a full DB write (wipe + bulk COPY) for this run. Semantic search ` +
`falls back to exact scan until VECTOR is available; run \`gitnexus doctor\` for ` +
`live extension status, or set GITNEXUS_LBUG_EXTENSION_INSTALL=auto to allow one ` +
`bounded install attempt.`
: `Incremental: effective write set covers ${effectiveWriteSet.size}/${allFilePaths.length} ` +
// Display clamp only (predicate unchanged): BFS-found deleted
// importers can push the numerator past the CURRENT file list, so
// the raw fraction can exceed 1 — see the population-mismatch note
// on shouldEscalateIncrementalWrite (tri-review 4669518496).
`files (${Math.min(100, Math.round(writeFraction * 100))}%) — switching to a full DB write ` +
`(wipe + bulk COPY) for this run; file-level incremental bookkeeping is unaffected.`,
`Incremental: effective write set covers ${effectiveWriteSet.size}/${allFilePaths.length} ` +
// Display clamp only (predicate unchanged): BFS-found deleted
// importers can push the numerator past the CURRENT file list, so
// the raw fraction can exceed 1 — see the population-mismatch note
// on shouldEscalateIncrementalWrite (tri-review 4669518496).
`files (${Math.min(100, Math.round(writeFraction * 100))}%) — switching to a full DB write ` +
`(wipe + bulk COPY) for this run; file-level incremental bookkeeping is unaffected.`,
);
// toWriteCount: 0 is the established full-path dirty-flag sentinel;
// the real counters ride along for crash diagnostics.
@@ -1767,8 +1636,8 @@ export async function runFullAnalysis(
// to replace wholesale.
await walCheckpointDriver.stop();
await closeLbug();
await wipeLbugDbFiles(buildPath);
await initLbug(buildPath);
await wipeLbugDbFiles(lbugPath);
await initLbug(lbugPath);
walCheckpointDriver = startWalCheckpointDriver();
await loadGraphToLbug(pipelineResult.graph, pipelineResult.repoPath, storagePath, (msg) => {
lbugMsgCount++;
@@ -1776,20 +1645,7 @@ export async function runFullAnalysis(
progress('lbug', pct, msg);
});
} else {
// 1a. Drop every FTS index before touching a single row (#2589).
// `deleteNodesForFiles` below DETACH DELETEs rows out of tables
// that otherwise still carry the FTS index built at the end of
// the PREVIOUS analyze run — Phase 3 doesn't drop+rebuild it
// until well after this delete completes. LadybugDB's FTS
// extension is not proven to survive DML against an indexed
// table (its own docs never demonstrate it), and that ordering
// is exactly what produced "FTS index 'file_fts' is
// inconsistent: term is missing during delete". Dropping first
// removes the hazard outright; Phase 3's createSearchFTSIndexes
// rebuilds every index from the final row set regardless, so
// this is a no-op on its own drop step there.
await dropSearchFTSIndexes();
// 1b. Remove the write set's existing rows — batched (#2409): one
// 1a. Remove the write set's existing rows — batched (#2409): one
// DETACH DELETE per table per 200-file chunk. The former per-file
// loop issued a count + delete per table per FILE — ~13k
// single-row write transactions on a ~700-file write set — which
@@ -2112,8 +1968,8 @@ export async function runFullAnalysis(
// the case a naive gate would leave index-less again.
// buildVectorIndex carries its own extension-policy gate and
// warn-on-failure; the boolean feeds semanticMode so the finalize stamp
// reflects the DB's ACTUAL state even when recreation fails (extension
// unavailable → 'exact-scan').
// reflects the DB's ACTUAL state even when recreation fails (win32 /
// extension unavailable → 'exact-scan').
const dbWasWiped = !isIncremental || escalatedFullWrite;
if (restoredEmbeddingCount > 0 && dbWasWiped && embeddingSkipped) {
// Re-import at the seam rather than thread a mutable capture from
@@ -2369,11 +2225,7 @@ export async function runFullAnalysis(
// inside the resolver, and a mismatch leaves the dirty flag intact so the
// next run takes the established full-recovery path.
meta.runnerIdentity = finalizeAnalyzerRunnerIdentity(import.meta.url, runnerIdentity);
// #2614 F1: the freshness stamp (saveMeta) is written AFTER the atomic swap
// below — never here — so a concurrent MCP reader can't observe
// meta.indexedAt = T_new while lbugPath still resolves to the pre-swap
// inode (which latched the reader on the stale index permanently). The meta
// object is fully computed at this point; only its write is deferred.
await saveMeta(metaDir, meta);
// Persist the incremental parse cache for the next run. Wraps in
// try/catch so a cache-write failure never breaks an otherwise
@@ -2511,59 +2363,7 @@ export async function runFullAnalysis(
// LadybugDB destructor double-free after --pdg writes — closeLbugBeforeExit
// CHECKPOINTs for durability then leaves the handles for process exit to
// reclaim (#2264). Long-lived callers close for real.
//
// On Windows a swap must release the build handle before the rename (a
// same-process open file can't be renamed), so it forces a real close —
// safe because windowsSwapOk excludes --pdg (the #2264 case). POSIX renames
// an open file, so it keeps the skip-native-close there.
const forceRealCloseForSwap = useAtomicSwap && process.platform === 'win32';
await (options.skipNativeCloseOnExit && !forceRealCloseForSwap
? closeLbugBeforeExit()
: closeLbug());
// #2 atomic publish: the fresh index was built at buildPath (a full rebuild,
// or an opt-in atomic incremental that copied the live index in first). Swap
// it over the live lbugPath in one rename so an MCP reader that opened
// mid-build only ever saw the previous complete index — never a wiped/
// half-built file. The close above checkpoint-consolidated buildPath to a
// single file (no .wal), so the rename publishes a complete index; a reader
// holding the old inode keeps a consistent stale snapshot until the pool
// re-opens onto the new one (the pool staleness invalidation). Runs only on
// success — a thrown error skips this, leaving the live index intact and the
// temp build to be cleared by the next run's wipe.
// Only publish if the build actually produced a DB at buildPath. A
// degenerate run (empty repo, or a mocked pipeline that never opened the
// store) leaves nothing to swap — skip rather than throw ENOENT.
const builtDbExists = useAtomicSwap
? await fs.stat(buildPath).then(
() => true,
() => false,
)
: false;
if (useAtomicSwap && builtDbExists) {
await retryRename(buildPath, lbugPath);
// Clear any sidecars orphaned beside the replaced file. A cleanly-closed
// prior index has none; a crashed one could, and it would be replay
// poison next to the freshly published index. Best-effort.
for (const suffix of ['.wal', '.shadow', '.wal.checkpoint'] as const) {
await fs.rm(`${lbugPath}${suffix}`, { force: true }).catch(() => {});
}
// #2614 F4: if the final checkpoint silently failed, the build may still
// carry a residual .wal/.shadow under the temp name. MOVE it beside the
// published index (not orphan/delete it) so the next open replays the
// delta, rather than leaving it under a name LadybugDB never reconciles.
for (const suffix of ['.wal', '.shadow'] as const) {
await fs.rename(`${buildPath}${suffix}`, `${lbugPath}${suffix}`).catch(() => {});
}
}
// #2614 F1: stamp the freshness metadata now that the index is published.
// When meta.indexedAt becomes visible, lbugPath already resolves to the new
// inode, so a reader reiniting on the stamp opens the fresh graph rather
// than latching on the old one. Leaving the dirty flag set across the swap
// is a crash-safety improvement: a failed swap leaves the previous index
// live and the next run recovers via the full-rebuild path.
await saveMeta(metaDir, meta);
await (options.skipNativeCloseOnExit ? closeLbugBeforeExit() : closeLbug());
progress('done', 100, 'Done');
-14
View File
@@ -121,20 +121,6 @@ export function getSearchFTSStemmer(): string {
return resolvedStemmer ?? resolveFTSStemmer();
}
/**
* Drop every configured FTS index (no-op per index when absent or unloadable
* — `dropFTSIndex` tolerates both). Callable ahead of any DML that mutates an
* FTS-indexed table's rows: LadybugDB's FTS extension is not proven to
* survive a DETACH DELETE against a table that still carries a live index
* from a prior run (#2589) — dropping first removes that hazard entirely,
* regardless of whether it also fixed a specific native inconsistency.
*/
export async function dropSearchFTSIndexes(): Promise<void> {
for (const { table, indexName } of FTS_INDEXES) {
await dropFTSIndex(table, indexName);
}
}
export async function createSearchFTSIndexes(
options?: CreateSearchFTSIndexesOptions,
): Promise<void> {
+145 -163
View File
@@ -15,8 +15,6 @@ import {
executeParameterized,
closeLbug,
isLbugReady,
statDbIdentity,
dbIdentityChanged,
} from '../../core/lbug/pool-adapter.js';
import { queryClassBeanMetadata } from './bean-metadata.js';
import { isValidQueryParams } from '../../core/lbug/query-params.js';
@@ -61,7 +59,10 @@ import {
type ExactEmbeddingRow,
} from '../../core/embeddings/exact-search.js';
import { EMBEDDING_TABLE_NAME, EMBEDDING_INDEX_NAME } from '../../core/lbug/schema.js';
import { getExactScanLimit } from '../../core/platform/capabilities.js';
import {
getExactScanLimit,
isVectorExtensionSupportedByPlatform,
} from '../../core/platform/capabilities.js';
import { PhaseTimer } from '../../core/search/phase-timer.js';
import { ftsDegradedWarning } from '../../core/search/fts-indexes.js';
import {
@@ -724,12 +725,6 @@ export class LocalBackend {
// not persist across calls and the staleness check would reinit forever
// (#2106).
private lastObservedIndexedAt: Map<string, string> = new Map();
// #2614 F1: file identity of the lbug the pool last opened. An atomic swap or
// an in-place incremental changes the inode; reiniting on that reinit-covers
// the window where meta.indexedAt hasn't caught up (and the incremental case),
// so a rebuilt index is never served stale even when the stamp looks current.
private lastObservedDbIdentity: Map<string, Awaited<ReturnType<typeof statDbIdentity>>> =
new Map();
private groupToolSvc: GroupService | null = null;
/**
* One-shot stderr warnings for sibling-clone drift, keyed by
@@ -1065,7 +1060,6 @@ export class LocalBackend {
this.initializedRepos.delete(key);
this.lastStalenessCheck.delete(key);
this.lastObservedIndexedAt.delete(key);
this.lastObservedDbIdentity.delete(key);
this.reinitPromises.delete(key);
closeLbug(key).catch(() => {});
}
@@ -1469,40 +1463,22 @@ export class LocalBackend {
// Reading the flat meta for a branch handle would compare the branch
// index's indexedAt against the primary's and thrash the pool (#2106).
const meta = await loadMeta(path.dirname(repo.lbugPath));
if (!meta) return;
// Compare against the last indexedAt OBSERVED for this pool (keyed by
// lbugPath), not the handle's — branch handles are fresh spreads so a
// handle mutation would not persist and would reinit on every check.
const observed = this.lastObservedIndexedAt.get(poolKey) ?? repo.indexedAt;
const stampChanged = !!meta?.indexedAt && meta.indexedAt !== observed;
// #2614 F1: also reinit on a file-identity change. An atomic swap (or an
// in-place incremental) changes the lbug inode; keying only on
// meta.indexedAt let a reader that reinited inside the pre-swap window
// latch on the old inode forever (its stamp already == meta.indexedAt).
const currentIdentity = await statDbIdentity(repo.lbugPath);
const identityChanged = dbIdentityChanged(
this.lastObservedDbIdentity.get(poolKey) ?? null,
currentIdentity,
);
if (stampChanged || identityChanged) {
// Index was rebuilt/swapped — DELEGATE the close/reopen to the pool's
// initLbug, which refuses to evict (and close the shared Database)
// while a query is in flight (its checkedOut>0 guard). Calling
// closeLbug directly here bypassed that guard and could close a
// Database mid-query — a native use-after-free (#2614). Wrap in
// reinitPromises to serialize concurrent detectors.
if (meta.indexedAt && meta.indexedAt !== observed) {
// Index was rebuilt — close stale connection and re-init.
// Wrap in reinitPromises to prevent TOCTOU race where concurrent
// callers both detect staleness and double-close the pool.
const reinit = (async () => {
try {
// Advance the observed stamp regardless: a stamp change with an
// unchanged file must not re-trigger on every check.
if (meta?.indexedAt) this.lastObservedIndexedAt.set(poolKey, meta.indexedAt);
const reopened = await initLbug(poolKey, repo.lbugPath);
// Advance the observed IDENTITY only when the pool actually rolled
// over. If a query was in flight, initLbug served the current
// handle and returned false; leaving the identity divergent
// re-triggers the reopen on a later idle check instead of latching.
if (reopened) {
this.lastObservedDbIdentity.set(poolKey, await statDbIdentity(repo.lbugPath));
}
await closeLbug(poolKey);
this.initializedRepos.delete(poolKey);
this.lastObservedIndexedAt.set(poolKey, meta.indexedAt);
await initLbug(poolKey, repo.lbugPath);
this.initializedRepos.add(poolKey);
} finally {
this.reinitPromises.delete(poolKey);
}
@@ -1521,7 +1497,6 @@ export class LocalBackend {
await initLbug(poolKey, repo.lbugPath);
this.initializedRepos.add(poolKey);
this.lastObservedIndexedAt.set(poolKey, repo.indexedAt);
this.lastObservedDbIdentity.set(poolKey, await statDbIdentity(repo.lbugPath));
} catch (err: any) {
// If lock error, mark as not initialized so next call retries
this.initializedRepos.delete(poolKey);
@@ -2416,16 +2391,10 @@ export class LocalBackend {
string,
{ distance: number; chunkIndex: number; startLine: number; endLine: number }
>();
// Always TRY the vector lane — no platform gate. LadybugDB ships the
// VECTOR extension for every supported platform, Windows included
// (#2623 follow-up; the old `platform !== 'win32'` gate was stale), so
// whether the index is queryable is a per-machine runtime fact. The
// catch below is the fallback: any failure (extension unloadable, index
// absent, older DB) degrades to the exact scan with a once-per-backend
// diagnostic instead of being silently swallowed.
try {
bestChunks = await collectBestChunks(limit, async (fetchLimit) => {
const vectorQuery = `
if (isVectorExtensionSupportedByPlatform()) {
try {
bestChunks = await collectBestChunks(limit, async (fetchLimit) => {
const vectorQuery = `
CALL QUERY_VECTOR_INDEX('${EMBEDDING_TABLE_NAME}', '${EMBEDDING_INDEX_NAME}',
CAST(${queryVecStr} AS FLOAT[${dims}]), ${fetchLimit})
YIELD node AS emb, distance
@@ -2436,27 +2405,27 @@ export class LocalBackend {
ORDER BY distance
`;
const embResults = await executeQuery(repo.lbugPath, vectorQuery);
return embResults.map((row) => ({
nodeId: row.nodeId ?? row[0],
chunkIndex: row.chunkIndex ?? row[1] ?? 0,
startLine: row.startLine ?? row[2] ?? 0,
endLine: row.endLine ?? row[3] ?? 0,
distance: row.distance ?? row[4],
}));
});
} catch (err) {
bestChunks = new Map();
if (!this.warnedVectorUnsupported) {
// Rare diagnostic: surface why semantic search fell back to the
// exact scan. Emitted once per `LocalBackend` instance lifetime to
// avoid noisy stderr on hot semantic-search paths (DoD §2.8).
this.warnedVectorUnsupported = true;
logger.warn(
{ err },
'GitNexus [query:vector]: vector index query failed; using exact scan fallback',
);
const embResults = await executeQuery(repo.lbugPath, vectorQuery);
return embResults.map((row) => ({
nodeId: row.nodeId ?? row[0],
chunkIndex: row.chunkIndex ?? row[1] ?? 0,
startLine: row.startLine ?? row[2] ?? 0,
endLine: row.endLine ?? row[3] ?? 0,
distance: row.distance ?? row[4],
}));
});
} catch {
bestChunks = new Map();
}
} else if (!this.warnedVectorUnsupported) {
// Rare diagnostic: surface why we fell back to the exact scan path so
// operators can see at a glance that VECTOR is disabled by platform
// policy. Emitted once per `LocalBackend` instance lifetime to avoid
// noisy stderr on hot semantic-search paths (DoD §2.8).
this.warnedVectorUnsupported = true;
logger.warn(
'GitNexus [query:vector]: VECTOR extension not supported on this platform; using exact scan fallback',
);
}
if (bestChunks.size === 0) {
@@ -4392,31 +4361,44 @@ export class LocalBackend {
return { error: 'New name is the same as the current name.' };
}
// Steps 2+3: Determine the set of files the apply step will rewrite, then
// enumerate every occurrence in each. The apply step (Step 4) does a
// whole-file `\boldName\b` global replace on every file in `changes`, so the
// reported edit list MUST enumerate every matching line in every such file —
// otherwise the preview under-reports what lands, and the same partial list
// comes back after apply (#2605). Building `changes` from one file set makes
// the preview enumerate exactly the files the apply loop rewrites, using the
// same word-boundary regex. (This is per-call consistency; the apply loop
// still re-reads each file, so an external write landing between preview and
// apply is a pre-existing gap this method does not lock against.)
type RenameEdit = {
line: number;
old_text: string;
new_text: string;
confidence: 'graph' | 'text_search';
// Step 2: Collect edits from graph (high confidence)
const changes = new Map<string, { file_path: string; edits: any[] }>();
const addEdit = (
filePath: string,
line: number,
oldText: string,
newText: string,
confidence: string,
) => {
if (!changes.has(filePath)) {
changes.set(filePath, { file_path: filePath, edits: [] });
}
changes.get(filePath)!.edits.push({ line, old_text: oldText, new_text: newText, confidence });
};
const escapedOldName = oldName.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
// Classify each file to rewrite by how it was discovered. Definition and
// graph-ref files carry graph confidence; files found only by text search
// carry text_search confidence. A graph-classified file is never downgraded.
const fileConfidence = new Map<string, 'graph' | 'text_search'>();
if (sym.filePath) {
fileConfidence.set(sym.filePath, 'graph');
// The definition itself
if (sym.filePath && sym.startLine) {
try {
const content = await fs.readFile(assertSafePath(sym.filePath), 'utf-8');
const lines = content.split('\n');
const lineIdx = sym.startLine - 1;
if (lineIdx >= 0 && lineIdx < lines.length && lines[lineIdx].includes(oldName)) {
const defRegex = new RegExp(
`\\b${oldName.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')}\\b`,
'g',
);
addEdit(
sym.filePath,
sym.startLine,
lines[lineIdx].trim(),
lines[lineIdx].replace(defRegex, new_name).trim(),
'graph',
);
}
} catch (e) {
logQueryError('rename:read-definition', e);
}
}
// All incoming refs from graph (callers, importers, etc.)
@@ -4426,13 +4408,44 @@ export class LocalBackend {
...(lookupResult.incoming.extends || []),
...(lookupResult.incoming.implements || []),
];
let graphEdits = changes.size > 0 ? 1 : 0; // count definition edit
for (const ref of allIncoming) {
if (ref.filePath) {
fileConfidence.set(ref.filePath, 'graph');
if (!ref.filePath) continue;
try {
const content = await fs.readFile(assertSafePath(ref.filePath), 'utf-8');
const lines = content.split('\n');
for (let i = 0; i < lines.length; i++) {
if (lines[i].includes(oldName)) {
addEdit(
ref.filePath,
i + 1,
lines[i].trim(),
lines[i]
.replace(
new RegExp(`\\b${oldName.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')}\\b`, 'g'),
new_name,
)
.trim(),
'graph',
);
graphEdits++;
break; // one edit per file from graph refs
}
}
} catch (e) {
logQueryError('rename:read-ref', e);
}
}
// Text search for files the graph might have missed entirely.
// Step 3: Text search for refs the graph might have missed
let astSearchEdits = 0;
const graphFiles = new Set(
[sym.filePath, ...allIncoming.map((r) => r.filePath)].filter(Boolean),
);
// Simple text search across the repo for the old name (in files not already covered by graph)
try {
const { execFileSync } = await import('child_process');
const rgArgs = [
@@ -4459,98 +4472,67 @@ export class LocalBackend {
for (const file of files) {
const normalizedFile = file.replace(/\\/g, '/').replace(/^\.\//, '');
// Never downgrade a graph-classified file to text_search.
if (!fileConfidence.has(normalizedFile)) {
fileConfidence.set(normalizedFile, 'text_search');
if (graphFiles.has(normalizedFile)) continue; // already covered by graph
try {
const content = await fs.readFile(assertSafePath(normalizedFile), 'utf-8');
const lines = content.split('\n');
const regex = new RegExp(`\\b${oldName.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')}\\b`, 'g');
for (let i = 0; i < lines.length; i++) {
regex.lastIndex = 0;
if (regex.test(lines[i])) {
regex.lastIndex = 0;
addEdit(
normalizedFile,
i + 1,
lines[i].trim(),
lines[i].replace(regex, new_name).trim(),
'text_search',
);
astSearchEdits++;
}
}
} catch (e) {
logQueryError('rename:text-search-read', e);
}
}
} catch (e) {
logQueryError('rename:ripgrep', e);
}
// Enumerate every `\boldName\b` line in each file to rewrite, so the previewed
// file set is exactly the set the apply loop below rewrites. A file with no
// matching line is dropped (apply would write nothing to it). `wordTest`
// (non-global) probes each line; `wordReplace` (global) rewrites it and is
// reused by the apply loop — compiled once each rather than once per line,
// and one escaping formula serves both passes.
const wordTest = new RegExp(`\\b${escapedOldName}\\b`);
const wordReplace = new RegExp(`\\b${escapedOldName}\\b`, 'g');
const changes = new Map<string, { file_path: string; edits: RenameEdit[] }>();
// Step 4: Apply or preview
const allChanges = Array.from(changes.values());
const totalEdits = allChanges.reduce((sum, c) => sum + c.edits.length, 0);
for (const [filePath, confidence] of fileConfidence) {
try {
const content = await fs.readFile(assertSafePath(filePath), 'utf-8');
const lines = content.split('\n');
const edits: RenameEdit[] = [];
for (let i = 0; i < lines.length; i++) {
if (!wordTest.test(lines[i])) {
continue;
}
edits.push({
line: i + 1,
old_text: lines[i].trim(),
new_text: lines[i].replace(wordReplace, new_name).trim(),
confidence,
});
}
if (edits.length > 0) {
changes.set(filePath, { file_path: filePath, edits });
}
} catch (e) {
logQueryError('rename:enumerate', e);
}
}
// Step 4: Apply or preview.
const failedFiles: string[] = [];
if (!dry_run) {
for (const change of changes.values()) {
// Apply edits to files
for (const change of allChanges) {
try {
const fullPath = assertSafePath(change.file_path);
const content = await fs.readFile(fullPath, 'utf-8');
await fs.writeFile(fullPath, content.replace(wordReplace, new_name), 'utf-8');
let content = await fs.readFile(fullPath, 'utf-8');
const regex = new RegExp(`\\b${oldName.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')}\\b`, 'g');
content = content.replace(regex, new_name);
await fs.writeFile(fullPath, content, 'utf-8');
} catch (e) {
// A swallowed write failure must not be reported as success (#2283):
// record the file so the result degrades to 'partial'.
// A swallowed write failure must not be reported as a full success
// (#2283): record the file so the result can degrade to 'partial'
// with the unwritten files listed, rather than masquerading as done.
logQueryError('rename:apply-edit', e);
failedFiles.push(change.file_path);
}
}
// A file whose write threw did not land, so drop its edits from the
// reported result — total_edits/changes must describe what actually
// reached disk, not what was attempted (#2605: the report matches reality
// even on partial failure). failed_files still names every dropped file.
for (const f of failedFiles) {
changes.delete(f);
}
}
// Counts derive from the reported set (dry-run: every enumerated file;
// apply: only files that landed), so the graph/text_search split always
// sums to total_edits and never overstates a partial apply.
const reported = Array.from(changes.values());
let graphEdits = 0;
let astSearchEdits = 0;
for (const change of reported) {
for (const edit of change.edits) {
if (edit.confidence === 'graph') {
graphEdits++;
} else {
astSearchEdits++;
}
}
}
return {
status: failedFiles.length > 0 ? 'partial' : 'success',
old_name: oldName,
new_name,
files_affected: reported.length,
total_edits: graphEdits + astSearchEdits,
files_affected: allChanges.length,
total_edits: totalEdits,
graph_edits: graphEdits,
text_search_edits: astSearchEdits,
changes: reported,
changes: allChanges,
applied: !dry_run,
...(failedFiles.length > 0 && { failed_files: failedFiles }),
};
-79
View File
@@ -1,7 +1,6 @@
import { execFileSync, execSync } from 'child_process';
import { statSync } from 'fs';
import path from 'path';
import os from 'os';
// Git utilities for repository detection, commit tracking, and diff analysis
@@ -210,84 +209,6 @@ export const getCanonicalRepoRoot = (fromPath: string): string | null => {
}
};
// getGitInfoExcludePath/getCoreExcludesFilePath are called once per repo
// PER language/contract extractor during group sync (#2606) — an N-repo
// group fans out to 6+ extractors each calling these, so an uncached
// execSync per call turns into O(extractors × repos) blocking subprocess
// spawns. Both resolve to the same value for the same fromPath for the
// life of the process (git config/exclude files don't change mid-run), so
// memoize by fromPath. ponytail: process-lifetime cache, never invalidated
// — fine for one-shot CLI runs; the long-lived MCP server would need a
// TTL or explicit invalidation if a user edits core.excludesFile mid-session.
const gitInfoExcludePathCache = new Map<string, string | null>();
const coreExcludesFilePathCache = new Map<string, string>();
/**
* Path to the repo's `$GIT_COMMON_DIR/info/exclude` file — git's own
* per-repo, untracked exclude list (same tier as `.gitignore` in
* precedence, but never committed, so it works even when the caller has
* no write access to the repo's tracked content). Shared across every
* linked worktree of a repo, matching git's own resolution (#2606).
*
* Returns `null` when `fromPath` is not inside a git repository or `git`
* is unavailable; callers should treat that the same as "no file".
*/
export const getGitInfoExcludePath = (fromPath: string): string | null => {
const cached = gitInfoExcludePathCache.get(fromPath);
if (cached !== undefined) return cached;
let result: string | null;
try {
const commonDir = chompGitOutput(
execSync('git rev-parse --path-format=absolute --git-common-dir', {
cwd: fromPath,
stdio: ['ignore', 'pipe', 'ignore'],
windowsHide: true,
}),
);
result = commonDir ? path.join(path.resolve(commonDir), 'info', 'exclude') : null;
} catch {
result = null;
}
gitInfoExcludePathCache.set(fromPath, result);
return result;
};
/**
* Path to git's own global, all-repos ignore file: the value of
* `core.excludesFile` (any config scope — system/global/local, resolved
* the same way `git` itself would from `fromPath`), or git's documented
* default of `$XDG_CONFIG_HOME/git/ignore` when unset (gitignore(5)).
* Lowest-precedence source, mirroring git's own behavior (#2606).
*
* Never throws: an unset key or unavailable `git` falls through to the
* default path, which is always computable without `git`.
*/
export const getCoreExcludesFilePath = (fromPath: string): string => {
const cached = coreExcludesFilePathCache.get(fromPath);
if (cached !== undefined) return cached;
let result: string | undefined;
try {
const configured = chompGitOutput(
execSync('git config --get --type=path core.excludesFile', {
cwd: fromPath,
stdio: ['ignore', 'pipe', 'ignore'],
windowsHide: true,
}),
);
if (configured) result = configured;
} catch {
// Unset, or git unavailable — fall through to git's documented default.
}
if (!result) {
const xdgConfigHome = process.env.XDG_CONFIG_HOME || path.join(os.homedir(), '.config');
result = path.join(xdgConfigHome, 'git', 'ignore');
}
coreExcludesFilePathCache.set(fromPath, result);
return result;
};
/**
* Resolve `fromPath` to the directory whose basename should drive the
* registry name (#1259) — the *identity root*. Three outcomes:
+1 -16
View File
@@ -429,23 +429,8 @@ export interface RepoMeta {
* `E.hook` to `E$1.hook`, and nested-host anonymous names re-key
* (`EnumWrap$1` → `EnumWrap$Mode$1`). Same contract as v8: identities move
* on unchanged files; force a full re-analyze.
* v10: Java `record_declaration` now emits a first-class `Record` graph node
* (#2564): a record's container node was previously never created (JAVA_QUERIES
* had no capture for it), so its methods existed as ownerless Method nodes
* with no `HAS_METHOD` edge. The incremental write set only covers changed
* files — a top-up against a pre-v10 index would keep silently omitting the
* `Record` node and its `HAS_METHOD` edges for every unchanged record file
* (same v7 contract: new nodes/edges the incremental path would otherwise
* never backfill); force a full re-analyze instead.
* v11: Rust abstract trait methods (`fn foo(&self) -> T;`, no body) now get a
* scope + declaration capture (#2604): RUST_SCOPE_QUERY had no
* `function_signature_item` pattern, so a `&dyn Trait` receiver could never
* dispatch a CALLS edge to the trait's own method. Same v7/v10 contract: the
* incremental write set only covers changed files, so a top-up against a
* pre-v11 index would keep silently missing these CALLS edges for every
* unchanged Rust trait file; force a full re-analyze instead.
*/
export const INCREMENTAL_SCHEMA_VERSION = 11;
export const INCREMENTAL_SCHEMA_VERSION = 9;
export interface IndexedRepo {
repoPath: string;
@@ -21,12 +21,4 @@ class Unrelated {
public void caller() {
hook();
}
public void dispatchToConstant() {
EnumConst.A.hook();
}
public void dispatchInherited() {
EnumConst.A.log();
}
}
@@ -1,13 +0,0 @@
public enum Plain {
A;
public void m() {
System.out.println("plain m");
}
}
class PlainCaller {
public void callPlain() {
Plain.A.m();
}
}
@@ -1,12 +0,0 @@
package probe;
public class LocalChain {
void m() {
class Local {
void inner() {
System.out.println("right target");
}
}
new Local().inner();
}
}
@@ -1,7 +0,0 @@
package probe;
class Other {
void inner() {
System.out.println("wrong target");
}
}
@@ -1,11 +0,0 @@
package probe;
public record Point(int x, int y) {
public int sum() {
return x + y;
}
public int scaled(int factor) {
return sum() * factor;
}
}
@@ -1,3 +0,0 @@
class KnowledgeGraphService:
def extract_and_store_graph(self, text: str) -> None:
pass
@@ -1,23 +0,0 @@
from knowledge_graph_service import KnowledgeGraphService
class MemoryService:
def __init__(self, knowledge_graph_service: KnowledgeGraphService):
self.knowledge_graph_service = knowledge_graph_service
def store_memory(self, text: str) -> None:
self.knowledge_graph_service.extract_and_store_graph(text)
def archive_memory(self, text: str) -> None:
self.knowledge_graph_service.extract_and_store_graph(text)
def restore_memory(self, text: str) -> None:
self.knowledge_graph_service.extract_and_store_graph(text)
class ExplicitFieldMemoryService:
def __init__(self, knowledge_graph_service):
self.knowledge_graph_service: KnowledgeGraphService = knowledge_graph_service
def ingest_memory(self, text: str) -> None:
self.knowledge_graph_service.extract_and_store_graph(text)
@@ -1,6 +0,0 @@
def extract_and_store_graph(text: str) -> None:
pass
def exercise_decoy(text: str) -> None:
extract_and_store_graph(text)
@@ -1,15 +0,0 @@
pub trait Behaviour {
fn trait_target(&self) -> u32;
}
pub struct Impl1;
impl Behaviour for Impl1 {
fn trait_target(&self) -> u32 {
7
}
}
pub fn calls_via_dyn(b: &dyn Behaviour) -> u32 {
b.trait_target()
}
@@ -88,8 +88,8 @@
"digest": "338c3922981604e71ddfc60ad61eba4b17f68ca654644e01add942c729b422cf"
},
"python-call-result-binding/models.py": {
"captureGroups": 17,
"digest": "441e2596001c4eaea4808ae6dd195a031f99d8ffe38c8fd450415c109d3365e2"
"captureGroups": 16,
"digest": "cbbb5168c28123820a70fe24016b0ed26ab02a6339d21ffe40fbccad940c1d70"
},
"python-call-result-binding/service.py": {
"captureGroups": 9,
@@ -171,25 +171,13 @@
"captureGroups": 15,
"digest": "201ce01b83b21d729aca89c6299570df55393a95f1899a2eaacd989873950177"
},
"python-constructor-field-receiver/knowledge_graph_service.py": {
"captureGroups": 10,
"digest": "83dcf9f81ac7d0a9e9ed5e467e41acd07e2ec581926d85eea5d4b5e1e4157744"
},
"python-constructor-field-receiver/memory_service.py": {
"captureGroups": 59,
"digest": "ad9c3be5a7c10e112bb20eac20b8603196fc2e739b7567159c58a607639ce192"
},
"python-constructor-field-receiver/test_fixture.py": {
"captureGroups": 13,
"digest": "6f642e752086a5e9337d21accb65ca90ef5af2b3e139239ea12c5cd031844dd2"
},
"python-constructor-type-inference/models/repo.py": {
"captureGroups": 17,
"digest": "3c400c7a331d7796a730e1ba53b91c4f5ec4799121044b0c160844988fca8662"
"captureGroups": 16,
"digest": "ad11823ee187cc3e1efab34a67a5013119b4ada87c5701b080921c0b0be09e62"
},
"python-constructor-type-inference/models/user.py": {
"captureGroups": 17,
"digest": "441e2596001c4eaea4808ae6dd195a031f99d8ffe38c8fd450415c109d3365e2"
"captureGroups": 16,
"digest": "cbbb5168c28123820a70fe24016b0ed26ab02a6339d21ffe40fbccad940c1d70"
},
"python-constructor-type-inference/services/app.py": {
"captureGroups": 13,
@@ -204,12 +192,12 @@
"digest": "dd51c32d705934b1384991ad2291869f446327752481abc20600d4ad9f553ea3"
},
"python-dict-items-loop/repo.py": {
"captureGroups": 16,
"digest": "2d283b4acbc71e318a4520cb7557508b9084213ba53fbdac07457e348a8b24c6"
"captureGroups": 15,
"digest": "8116cf4cbf4dca377e88f97ca645f40fab648a4e9a8e790b5b3761e4a3e17d7c"
},
"python-dict-items-loop/user.py": {
"captureGroups": 16,
"digest": "6568834a7f228e78a11b08282196e138795980aed6c55b9953cc89e87c998e52"
"captureGroups": 15,
"digest": "15984fa30be4603f3e47c27342352dd602d104b33ac5224dc911d78a82d87926"
},
"python-django-app-imports/accounts/__init__.py": {
"captureGroups": 0,
@@ -300,8 +288,8 @@
"digest": "d1e23831dcae38034b278bfefa2b8c4e21ca722ab2c79bfb3126744338a2a401"
},
"python-enumerate-loop/user.py": {
"captureGroups": 17,
"digest": "441e2596001c4eaea4808ae6dd195a031f99d8ffe38c8fd450415c109d3365e2"
"captureGroups": 16,
"digest": "cbbb5168c28123820a70fe24016b0ed26ab02a6339d21ffe40fbccad940c1d70"
},
"python-field-type-disambig/address.py": {
"captureGroups": 9,
@@ -328,8 +316,8 @@
"digest": "5c290b3b34f3f5e9dcdd6ee3ae72ba4223337cd64592330f9a8c6c6f13b2fd2d"
},
"python-for-call-expr/models.py": {
"captureGroups": 41,
"digest": "e8807a9969732197feb04204d200d5810b895c006f1424a7a6f5f0d5762ef49a"
"captureGroups": 39,
"digest": "133a14c0543a41412d9d4fd5485d6270e1f4b0cd247f0ec0d71974850e4918c1"
},
"python-function-local-import-chain/app.py": {
"captureGroups": 9,
@@ -476,8 +464,8 @@
"digest": "01fe4805f59723a5f163d26b7be3ed3e456eb3a093e3df3a8f034cecee22ebb9"
},
"python-method-chain-binding/models.py": {
"captureGroups": 48,
"digest": "f049817034428759193fce03b417ca23c133f6ead87f960e173afba9b79b88e8"
"captureGroups": 45,
"digest": "ea3f745514a330d86447796734faff29f0c88ea49a7ef8781087c537426d29bf"
},
"python-method-enrichment/app.py": {
"captureGroups": 13,
@@ -692,8 +680,8 @@
"digest": "741f690b6330491303b9b58cb31027a33600973265b59428facfefbabf0cf7e1"
},
"python-return-type-inference/models.py": {
"captureGroups": 17,
"digest": "441e2596001c4eaea4808ae6dd195a031f99d8ffe38c8fd450415c109d3365e2"
"captureGroups": 16,
"digest": "cbbb5168c28123820a70fe24016b0ed26ab02a6339d21ffe40fbccad940c1d70"
},
"python-return-type-inference/service.py": {
"captureGroups": 9,
@@ -772,8 +760,8 @@
"digest": "4df7ea089c43552ca4ea5a51f8e985d11d351b86949efb2f7ebefdf6a9ffd689"
},
"python-walrus-operator/models.py": {
"captureGroups": 22,
"digest": "9ab1a8a69970e8c875bd103251ecf9c8af2c1464a6bdc1213fe7560c4fa34063"
"captureGroups": 21,
"digest": "cf014d1bad66ea61e327c146fd1a52159a96faaf264555bb164230a5218e6948"
},
"python-write-access/models.py": {
"captureGroups": 11,
@@ -784,7 +772,7 @@
"digest": "6e3690ec68d8de54f376bb6f5f7a29da829a8a6a24001eabae9ec66a327f3409"
},
"synthetic:dao-20": {
"captureGroups": 773,
"digest": "37e047eda37477bbc33f4dd8ba259c3f876580378566c0c14dfe580795a952af"
"captureGroups": 733,
"digest": "c540f2143882137e6c6f7996bf9b87a0117f41915b1d0989d3d6bb9db5a5a1ab"
}
}
@@ -1,7 +1,7 @@
{
"rust-abstract-dispatch/src/lib.rs": {
"captureGroups": 34,
"digest": "973679363065ecd54c4e5128a9fab214ea27eca24f0f079c63a3c6f285e678b0"
"captureGroups": 30,
"digest": "88309004d1ab00054f81bc55c1d058fc4ca25781162d6b94b7e7ce631a5d61b2"
},
"rust-abstract-dispatch/src/main.rs": {
"captureGroups": 21,
@@ -148,8 +148,8 @@
"digest": "e0120e3f215282e68d83b4f8f5d8918945e0b3e7ce4e0128c6afd2aa43caa1c0"
},
"rust-cross-module-collision/src/traits.rs": {
"captureGroups": 5,
"digest": "c7150a5052e0e2b5fd7fc21cc8ce361530fe5cc60f67ad0e2ef945bb97965b53"
"captureGroups": 3,
"digest": "88eef9d92ea6e370bd8ef7fbf64c42ec622ec933fb53c56b32bc67db87fa8e03"
},
"rust-deep-field-chain/models.rs": {
"captureGroups": 24,
@@ -171,10 +171,6 @@
"captureGroups": 22,
"digest": "c53db401a81fde2ffd5665393acb9cd605a62ec51c015c3aafb3f41c0897471f"
},
"rust-dyn-trait-object/src/lib.rs": {
"captureGroups": 23,
"digest": "720618dff6a43ab8e5b59aa354c0c448b9057dd6f2f7b3b22b13b82d53745943"
},
"rust-err-unwrap/src/error.rs": {
"captureGroups": 9,
"digest": "798c8e01c6e54792ba69e845248efc8abf0cba38fa3d16fb8e0d1f6dd2ad2b7e"
@@ -320,8 +316,8 @@
"digest": "cd836a2a9c15ab240961d2e15f192f7e33d65eb5ebf2e1a8af2f620a47fe66ae"
},
"rust-method-enrichment/src/lib.rs": {
"captureGroups": 42,
"digest": "a4d9ca570fbb1ff1859a0b4f737aa3507b236f99700d37ded2c8c36518add567"
"captureGroups": 40,
"digest": "71627a8218e32514b6945e4e310686eb631ce37451c90c9644e3f5336a37820b"
},
"rust-method-enrichment/src/main.rs": {
"captureGroups": 18,
@@ -364,8 +360,8 @@
"digest": "141388068614e16d96f27cfdf18ac9001b9e202ce832fe10f38dab990637b3ab"
},
"rust-parent-resolution/src/serializable.rs": {
"captureGroups": 5,
"digest": "f33bb881dd937cdd5af2eca6b0284ea79ca2d296c79217513f882dcba8a82fd8"
"captureGroups": 3,
"digest": "f35d44f44d81e3a0be40f68ba9dbd4bde6f01659fa15b6db34a458ad460f904e"
},
"rust-parent-resolution/src/user.rs": {
"captureGroups": 13,
@@ -376,8 +372,8 @@
"digest": "bc8946d31db81b85d780633608fdaa7565258cd788285fa00cd6dcb0de3dd16c"
},
"rust-qualified-trait/src/traits.rs": {
"captureGroups": 9,
"digest": "10f3bba4c2a16cdac77de0498ff506910ac09cc1a85e7daebfc5743c54e26015"
"captureGroups": 5,
"digest": "15be069f28f1400e4beb0b0860acb59979f78549960486f36a92f56578f05a06"
},
"rust-qualified-trait/src/widget.rs": {
"captureGroups": 23,
@@ -496,12 +492,12 @@
"digest": "f1b9f72d74467be55a8b7679215b49bcabb4d0fced6080f752672070b32ed93d"
},
"rust-traits/src/traits/clickable.rs": {
"captureGroups": 7,
"digest": "83c36832f24446fe03a07288dd394bc7494a54e5979c9b71c1dcb8b19ab54341"
"captureGroups": 3,
"digest": "3ed5b27c172d48f83929715ba92d1030a282f9f1e29ec2fcdd3d7e9efbc54a84"
},
"rust-traits/src/traits/drawable.rs": {
"captureGroups": 9,
"digest": "cee5091f041038722f1f012394a75ba4e16870d05b2dafa37371200e785198f0"
"captureGroups": 5,
"digest": "1dca39bbc7c1b1b66f1a34730b9a5b4dba04c54ee9d2688255e0fd4e6bc48499"
},
"rust-union/lib.rs": {
"captureGroups": 10,
@@ -1,25 +0,0 @@
package com.example;
import org.springframework.beans.factory.annotation.Value;
import org.springframework.boot.context.properties.ConfigurationProperties;
class DirectValues {
@Value("${payment.timeout:30}")
private int timeout;
@Value("${payment.missing}")
private String missing;
}
@ConfigurationProperties(prefix = "service")
class ServiceProperties {
private String endpoint;
private Retry retry;
}
@ConfigurationProperties("service")
class UnmatchedServiceProperties {
private String unrelated;
}
class Retry {}
@@ -1,6 +0,0 @@
defaults: &defaults
retry:
max-attempts: 3
service:
<<: *defaults
endpoint: https://service.example.test

Some files were not shown because too many files have changed in this diff Show More